mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 23:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ae34b1e521 | ||
|
|
c17da25617 | ||
|
|
cc8ef16240 | ||
|
|
30fac81c41 | ||
|
|
bbcca80b60 | ||
|
|
56ae3716bb | ||
|
|
2fba3a4a9a | ||
|
|
8ba0312d40 | ||
|
|
53623ffa72 | ||
|
|
d6199687f4 | ||
|
|
7cb413dde6 | ||
|
|
c5a7fc1e2a | ||
|
|
0869b0aa29 | ||
|
|
4e81f847f2 | ||
|
|
8fa27d2d9f | ||
|
|
e5a8a29efe | ||
|
|
f967f53176 | ||
|
|
31fefaae0d | ||
|
|
7f9edbf39e | ||
|
|
760b9c8e7f | ||
|
|
cc7fb008fc | ||
|
|
98dbfbe654 | ||
|
|
6444726614 | ||
|
|
0a562fbb10 | ||
|
|
7d8950de40 | ||
|
|
3fdde0c90b | ||
|
|
e776f4cd4e | ||
|
|
e7d7dd13d7 | ||
|
|
d5c83a2e45 | ||
|
|
37ccb13917 | ||
|
|
8439cf410f | ||
|
|
614d9448f7 | ||
|
|
70efdbba5d | ||
|
|
12fee91ebb | ||
|
|
bcb7e20619 | ||
|
|
afc5e8a140 | ||
|
|
4be6626c89 | ||
|
|
5f2b6d629b | ||
|
|
1b8f5f08f0 | ||
|
|
79674c697a | ||
|
|
416d8c1df6 | ||
|
|
d03b6a9382 | ||
|
|
df22e0c796 | ||
|
|
79a3bd75cc | ||
|
|
e6acdcc583 | ||
|
|
cd78984228 | ||
|
|
2f007c5f0a | ||
|
|
ae64a1e30c | ||
|
|
097184c3e0 | ||
|
|
84a95adae1 | ||
|
|
123b6728e4 | ||
|
|
69b4fc98ed | ||
|
|
35cf7703b1 | ||
|
|
8c1137543b | ||
|
|
b8e66e56a0 | ||
|
|
fbb008e510 | ||
|
|
04678f8404 | ||
|
|
1b5aaf1fb8 | ||
|
|
d8e4873f43 | ||
|
|
998a3d8353 | ||
|
|
336dedbed5 | ||
|
|
955595be8a | ||
|
|
576bd4f69a | ||
|
|
cd6915917b | ||
|
|
0adbe31112 | ||
|
|
d5138f509c | ||
|
|
1fe6fc3feb | ||
|
|
c03a7fd482 | ||
|
|
31c47f05f4 | ||
|
|
edad24479b | ||
|
|
bba58732e0 | ||
|
|
8d1cc6cf51 | ||
|
|
f03d0be5ef | ||
|
|
a2f4f494a9 | ||
|
|
9de25c200c | ||
|
|
fe1f00aadb | ||
|
|
a847fac4ca | ||
|
|
0496506fb7 | ||
|
|
e544591c9c | ||
|
|
e5237d1149 | ||
|
|
b9c848c5e9 | ||
|
|
b64e61a793 | ||
|
|
e80e2bdafe | ||
|
|
ac23bce0ba | ||
|
|
1051cd97cf | ||
|
|
45330fdd5d | ||
|
|
bd296d7a11 | ||
|
|
249e19bf2a | ||
|
|
baf52ed286 | ||
|
|
dd7e1baa78 | ||
|
|
80a209d6dc | ||
|
|
ab228c1fcb | ||
|
|
0dde233b1e | ||
|
|
cf0d92d968 | ||
|
|
ce4b052242 | ||
|
|
125bb3afe1 | ||
|
|
30ee2c18ef | ||
|
|
ff1a5dd4c2 | ||
|
|
b9c9c7d671 | ||
|
|
62d9961bd1 | ||
|
|
4b7ed9599d | ||
|
|
c8951de9dd | ||
|
|
63517c377d | ||
|
|
5005ebdfc9 | ||
|
|
a2615f0e52 | ||
|
|
0317d381fe | ||
|
|
31b5181bca | ||
|
|
7ae59055ff | ||
|
|
3524117c23 | ||
|
|
e8ad1ca0a0 | ||
|
|
3ac1650001 | ||
|
|
4f8ae81562 | ||
|
|
329d624a99 | ||
|
|
de92624c9a | ||
|
|
c6a034da40 | ||
|
|
737e76968a | ||
|
|
7c6d49155c | ||
|
|
4d404ea94d | ||
|
|
dc810c7a1e | ||
|
|
e0ee4e71f8 | ||
|
|
ab7dac90a2 | ||
|
|
3a423fbc41 | ||
|
|
0982ec617d | ||
|
|
b45f27c8d0 | ||
|
|
54f62b6701 | ||
|
|
9664d98ba5 | ||
|
|
0c318834b5 | ||
|
|
4c2836f4b3 | ||
|
|
a112db169c | ||
|
|
0dc33ec893 | ||
|
|
0e93ba532f | ||
|
|
9e524a30d6 | ||
|
|
ecca4e4abc | ||
|
|
51791e9efa | ||
|
|
5eab087cc2 | ||
|
|
cd05cdaa57 | ||
|
|
f5e18ccea6 | ||
|
|
0da6b07770 | ||
|
|
05805356fb | ||
|
|
a72ebfdec0 | ||
|
|
59e5da9c7f | ||
|
|
cfd59db998 | ||
|
|
1fa6bf1cc8 | ||
|
|
a2f89de71e | ||
|
|
36a27de286 | ||
|
|
9b685ba824 | ||
|
|
8e9d5fe6ac | ||
|
|
89aa590615 | ||
|
|
3960e0f2a3 | ||
|
|
c3c52d01d3 | ||
|
|
0caef59e6a | ||
|
|
99e6924ed1 | ||
|
|
431e1629b0 | ||
|
|
ff9f702e52 | ||
|
|
6903159f30 | ||
|
|
504b7a03ad | ||
|
|
6933c2aaff | ||
|
|
5fd00e31bb | ||
|
|
a984674dae | ||
|
|
8589119725 | ||
|
|
5e0205378b | ||
|
|
bff2f2e5f9 | ||
|
|
5f9052a675 | ||
|
|
8691b3964f | ||
|
|
4dfe0a0595 | ||
|
|
164299ce25 | ||
|
|
9269ca6ebe | ||
|
|
93fe22e045 | ||
|
|
3849278d5a | ||
|
|
c0f976e200 | ||
|
|
3143749f64 | ||
|
|
1704805e32 | ||
|
|
3aabe07720 | ||
|
|
c5815ab59e | ||
|
|
f6fcfbabd5 | ||
|
|
a0835c95e4 | ||
|
|
ce0c24eee9 | ||
|
|
c9f0ecb1e8 | ||
|
|
50febb3459 | ||
|
|
601b0f96b1 | ||
|
|
018661609a | ||
|
|
98d935d972 | ||
|
|
b07660c4fa | ||
|
|
8037231376 | ||
|
|
6e422a8691 | ||
|
|
3d347ed565 | ||
|
|
13446d8b6b | ||
|
|
08c34bf573 | ||
|
|
3a64ea1650 | ||
|
|
db3439b61b | ||
|
|
7814be7467 | ||
|
|
8a7e0432f1 | ||
|
|
ff1d51c7bd | ||
|
|
f2602e3136 | ||
|
|
313ce0fed8 | ||
|
|
ba0887defa | ||
|
|
063841f491 | ||
|
|
4aa5c78150 | ||
|
|
c1688fa392 | ||
|
|
0dd03e9cb6 | ||
|
|
8d60d70553 | ||
|
|
c17340547c | ||
|
|
34aa6bf6c7 | ||
|
|
82a6ce63b4 | ||
|
|
3ac6ba0fe2 | ||
|
|
8b0fb4fa19 | ||
|
|
9d437b8863 | ||
|
|
f72469c80a | ||
|
|
eec7972808 | ||
|
|
fc9343e7a4 | ||
|
|
be8e353d12 | ||
|
|
53217acdd0 | ||
|
|
c1e0ea5e0a | ||
|
|
467afc7248 | ||
|
|
62753071f4 | ||
|
|
de32de9edd | ||
|
|
aa7d2c3f7e | ||
|
|
2a1dd68583 | ||
|
|
80909eaa84 | ||
|
|
c6a557abe2 | ||
|
|
8d7b8729f1 | ||
|
|
3f9a1c3751 | ||
|
|
0dfe617d96 | ||
|
|
235ad441b4 | ||
|
|
a9d00b3f8d | ||
|
|
fb41ba172d | ||
|
|
aec5b21d2a | ||
|
|
124097d563 | ||
|
|
e137c2edff | ||
|
|
2b35dd829c | ||
|
|
686895802a | ||
|
|
72d0228cd7 | ||
|
|
298e6cad0c | ||
|
|
ad34c228e3 | ||
|
|
66f74a431f | ||
|
|
57bae90c69 | ||
|
|
83c2e52ba5 | ||
|
|
4f8525da4d | ||
|
|
3b8491b558 | ||
|
|
f69c53d294 | ||
|
|
6b226dd6af | ||
|
|
227462e4d8 | ||
|
|
5f00ba5cae | ||
|
|
0b8a6d9599 | ||
|
|
13bf04a81e | ||
|
|
7eb8409f71 | ||
|
|
e821072cb9 | ||
|
|
9c01dd9d9f | ||
|
|
6a43db8c8f | ||
|
|
3ffc301dd0 | ||
|
|
46fcbe2fc0 | ||
|
|
b7806e47e9 | ||
|
|
a1ed54adc7 | ||
|
|
250a4ea4e3 | ||
|
|
b3e090c8ff | ||
|
|
982518d3a4 | ||
|
|
9ad1d5548d | ||
|
|
0de2558ef7 | ||
|
|
3e0e601616 | ||
|
|
8b202b0ebf | ||
|
|
6d2f98a379 | ||
|
|
84379b5fdf | ||
|
|
d005fdcd03 | ||
|
|
a97fb2f34f | ||
|
|
d48981b6b0 | ||
|
|
af32228e38 | ||
|
|
8b35275ec1 | ||
|
|
302a6c96ff | ||
|
|
f4a4b2c14b | ||
|
|
88b94bef54 | ||
|
|
751b66d45d | ||
|
|
ae6a57e667 | ||
|
|
9110546d34 | ||
|
|
1f1d0706dd | ||
|
|
a82d41ee22 | ||
|
|
e8e70828d1 | ||
|
|
b7a58fdb8a | ||
|
|
dc9fde8fae | ||
|
|
c0cf4f6a61 | ||
|
|
21fc6bedcb | ||
|
|
ad6fd5ab72 | ||
|
|
0f696c6092 | ||
|
|
9af3bc1864 | ||
|
|
b020e593a5 | ||
|
|
4771a340f5 | ||
|
|
4449b60459 | ||
|
|
4139332ad9 | ||
|
|
30a28ff1ad | ||
|
|
498d0fc145 | ||
|
|
4a547fe95f | ||
|
|
ca906589d4 | ||
|
|
8bafae2262 | ||
|
|
4ef82c82b5 | ||
|
|
e03d253310 | ||
|
|
48c45da2de | ||
|
|
04a1ac967c | ||
|
|
aa17f64593 | ||
|
|
ae5cfcc249 | ||
|
|
8f578b57f2 | ||
|
|
3871646611 | ||
|
|
48a574dfd3 | ||
|
|
ec49100c63 | ||
|
|
c04d2409da | ||
|
|
b93b713179 | ||
|
|
fe2f54fc3d | ||
|
|
599f8f99ed | ||
|
|
b1b338ed5a | ||
|
|
e4d659a619 | ||
|
|
9ce94266e1 | ||
|
|
5a19425b28 | ||
|
|
1494aac861 | ||
|
|
8a21ecabee | ||
|
|
005b2bc3db | ||
|
|
f27830bf41 | ||
|
|
5fe6afc270 | ||
|
|
4f9bc70562 | ||
|
|
835155021d | ||
|
|
e2c5c2292d | ||
|
|
072690a191 | ||
|
|
a2c9d5a398 | ||
|
|
5944342604 | ||
|
|
4dbcd34548 | ||
|
|
f02c73a2c2 | ||
|
|
158ba1ae3b | ||
|
|
a85a77e088 | ||
|
|
542ab04671 | ||
|
|
28ee2ca5a2 | ||
|
|
3913dd6c8b | ||
|
|
1ecf147e3e | ||
|
|
d8fa53a445 | ||
|
|
1c1ad876ca | ||
|
|
5cd0f6e7e7 | ||
|
|
2627ed3193 | ||
|
|
e3bb8b43ab | ||
|
|
8ba6a3bda9 | ||
|
|
d288249609 | ||
|
|
33d7c6be9e | ||
|
|
c13acc5b60 | ||
|
|
fc9be28d51 | ||
|
|
944f400d19 | ||
|
|
5aee30a7b5 | ||
|
|
58ae49372c | ||
|
|
917e69b021 | ||
|
|
9242e59841 | ||
|
|
05d15fa052 | ||
|
|
0a62a4c571 | ||
|
|
525266e482 | ||
|
|
503881f5df | ||
|
|
04fd4f2bb4 | ||
|
|
8be1e5e260 | ||
|
|
e84e084d56 | ||
|
|
12ad05e089 | ||
|
|
472ce7c1e4 | ||
|
|
19fbc10c16 | ||
|
|
39f3406fba | ||
|
|
19b0a9cd7d | ||
|
|
eac579f714 | ||
|
|
b0a31f7105 | ||
|
|
29163cfd07 | ||
|
|
f05b24636f | ||
|
|
39466a3a8f | ||
|
|
bccf18f46c | ||
|
|
1109126e35 | ||
|
|
98dd40c3a8 | ||
|
|
b1cfb104ff | ||
|
|
c339cfe184 | ||
|
|
ca8e6a83e1 | ||
|
|
760217ab29 | ||
|
|
ee9361bf41 | ||
|
|
e62bc24b3f | ||
|
|
18074307f6 | ||
|
|
63b70ff3d4 | ||
|
|
dacdfd5c02 | ||
|
|
d5c039cdc7 | ||
|
|
e2e6f2a92b | ||
|
|
d7d8244593 | ||
|
|
b05e5ce14e | ||
|
|
08730c817a | ||
|
|
34e086eec2 | ||
|
|
83c8a9b675 | ||
|
|
a3db629fd6 | ||
|
|
64b3cef126 | ||
|
|
7100a2eaee | ||
|
|
ffcde1823b | ||
|
|
4c73c715ad | ||
|
|
26c3ebd6a7 | ||
|
|
790bd9747f | ||
|
|
83aa8731d1 | ||
|
|
bbd9eb5b9a | ||
|
|
cc90fa5773 | ||
|
|
30ebc6c938 | ||
|
|
c0a8984799 | ||
|
|
a6e34d301e | ||
|
|
82c88168cc | ||
|
|
d169c3ed4a | ||
|
|
563d11a702 | ||
|
|
31e2ba4696 | ||
|
|
2c3baaad1c | ||
|
|
01bd43aa2b | ||
|
|
f4d095ce0f | ||
|
|
8c5bd9a058 | ||
|
|
335ce1d056 | ||
|
|
0afb3caaed | ||
|
|
8521aaf65c | ||
|
|
0a451cfe0d | ||
|
|
cb0935c96b | ||
|
|
40ec910108 | ||
|
|
2d3c6efae2 | ||
|
|
c027acecf8 | ||
|
|
bc2840e4a7 | ||
|
|
869b472d91 | ||
|
|
81dfc21700 | ||
|
|
a3d8fe2362 | ||
|
|
d2a57f6231 | ||
|
|
9e9ceb3894 | ||
|
|
bd70088877 | ||
|
|
63c9d99b33 | ||
|
|
a2469f48e7 | ||
|
|
62f0f421a9 | ||
|
|
0b24758f41 | ||
|
|
147897e13d | ||
|
|
ade3a5275d | ||
|
|
51c5f945b4 | ||
|
|
a4e2e36243 | ||
|
|
2c5dc13201 | ||
|
|
43234939ca | ||
|
|
87b3d50899 | ||
|
|
da8dbf1777 | ||
|
|
34e1fcccf8 | ||
|
|
c99d1e48bd | ||
|
|
6a428043f6 | ||
|
|
aafe7ff10f | ||
|
|
952e157770 | ||
|
|
cae4f2f873 | ||
|
|
0fc6d6b6b5 | ||
|
|
7227ee9b2e | ||
|
|
c6153d6a52 | ||
|
|
d82d2944a9 | ||
|
|
58ad400519 | ||
|
|
d23c76d0a9 | ||
|
|
6a5b9e2a93 | ||
|
|
f33a93b0a1 | ||
|
|
015200f511 | ||
|
|
92f48819b6 | ||
|
|
0c6483cad5 | ||
|
|
ee02b1ca51 | ||
|
|
33845a3112 | ||
|
|
096ed29b5e | ||
|
|
3bbff8a948 | ||
|
|
726918b82c | ||
|
|
0f59a18223 | ||
|
|
3402cde334 | ||
|
|
8f53c6bb96 | ||
|
|
0d6e4631a3 | ||
|
|
8dd9a5bd38 | ||
|
|
903cf84874 | ||
|
|
e997da48c7 | ||
|
|
5ff89fd171 | ||
|
|
fad4254c0e | ||
|
|
2fb3c4f11c | ||
|
|
ce5297b75f | ||
|
|
b2b0c277f6 | ||
|
|
46919979ce | ||
|
|
8b716c6a22 | ||
|
|
75090f8f6c | ||
|
|
c14c0c2e3b | ||
|
|
95efd18b73 | ||
|
|
c9319a768f | ||
|
|
da48020882 | ||
|
|
ce4380e136 | ||
|
|
c633661121 | ||
|
|
4bfd1dde1f | ||
|
|
fe11bd2242 | ||
|
|
814f0c3c93 | ||
|
|
7e904056d3 | ||
|
|
969d8f866c | ||
|
|
a2d0b7d7c4 | ||
|
|
97a8fa77bc | ||
|
|
c1296cc64d | ||
|
|
1dee54a9d8 | ||
|
|
17e5d73e64 | ||
|
|
d523b7a6c7 | ||
|
|
24ad208778 | ||
|
|
fa87c73b9e | ||
|
|
0ed96544e1 | ||
|
|
d28ccc59ac | ||
|
|
eaa75c1ed2 | ||
|
|
4f4263263b | ||
|
|
ae00654694 | ||
|
|
a7156276e9 | ||
|
|
29859d2491 | ||
|
|
5460a24ea9 | ||
|
|
a284adcd19 | ||
|
|
73d43c1d55 | ||
|
|
256df76674 | ||
|
|
c8dc663b0b | ||
|
|
ba78dff1f8 | ||
|
|
1e597bfbed | ||
|
|
b78af2fdaf | ||
|
|
5379f0a9c7 | ||
|
|
f1f523e525 | ||
|
|
c79d79e08b | ||
|
|
ee2d417d21 | ||
|
|
cebdde599a | ||
|
|
c3ac72a01e | ||
|
|
13f3c6e75a | ||
|
|
b3cd4edb3b | ||
|
|
afe10c1666 | ||
|
|
d9d30916ba | ||
|
|
3f6c1c0e68 | ||
|
|
5b2cc77109 | ||
|
|
b824023ec6 | ||
|
|
6ce1be0880 | ||
|
|
c5dacab2ee | ||
|
|
4d24b85d57 | ||
|
|
e967b447e6 | ||
|
|
9bc631a427 | ||
|
|
c480ef137d | ||
|
|
15629e790e | ||
|
|
3f08d8b691 | ||
|
|
9d9d171aad | ||
|
|
65218c8285 | ||
|
|
27f2e0b06d | ||
|
|
ae75983b54 | ||
|
|
f8ba373e18 | ||
|
|
2feae06209 | ||
|
|
2e0534924a | ||
|
|
ad1fd7f54b | ||
|
|
9aaace51e1 | ||
|
|
58841142ee | ||
|
|
a6a816fb38 | ||
|
|
70988ccfee | ||
|
|
a8d9caf0c0 | ||
|
|
bc22186093 | ||
|
|
317416b2e0 | ||
|
|
b5ae9e4c97 | ||
|
|
099737ca05 | ||
|
|
e90164b519 | ||
|
|
933c1af7e8 | ||
|
|
b9d878b1f4 | ||
|
|
45a9a83c79 | ||
|
|
560cfc757c | ||
|
|
e92f51e415 | ||
|
|
278ca52d97 | ||
|
|
912dbfe5bd | ||
|
|
687f46fc71 | ||
|
|
6e9e5b3bd6 | ||
|
|
4fbc266b18 | ||
|
|
d1ac406895 | ||
|
|
ce0f5db6f7 | ||
|
|
efb42c1ad1 | ||
|
|
d8109880f4 | ||
|
|
09be28a443 | ||
|
|
fb0bb8dd2c | ||
|
|
8e36f5331f | ||
|
|
a365a70275 | ||
|
|
b5a4e5920d | ||
|
|
94d2ed85a7 | ||
|
|
90f338d7db | ||
|
|
df78f5d50e | ||
|
|
b2b4c2bdcf | ||
|
|
a888da436b | ||
|
|
3cb8ae9a9c | ||
|
|
db3854e391 | ||
|
|
05b4b095fe | ||
|
|
93926641d9 | ||
|
|
89d6752d3d | ||
|
|
e4f95fec79 | ||
|
|
8a7f39559c | ||
|
|
753d0ede6c | ||
|
|
da2e44d024 | ||
|
|
7b379fc3cf | ||
|
|
ec38d58b37 | ||
|
|
f6a74a710d | ||
|
|
3fd136b0da | ||
|
|
72e82d0304 | ||
|
|
2f7dcb8d93 | ||
|
|
128a24d699 | ||
|
|
cd94a8f0ac | ||
|
|
df5e0e5df9 | ||
|
|
458bbf4ef7 | ||
|
|
82319c9deb | ||
|
|
3bc4df7295 | ||
|
|
42a6320935 | ||
|
|
843fe378db | ||
|
|
3679673d5b | ||
|
|
3a269f04d2 | ||
|
|
0021723b50 | ||
|
|
1c4b0272e8 | ||
|
|
9640216124 | ||
|
|
0f8f2bf2b4 | ||
|
|
ef899b7b1a | ||
|
|
b85abf725d | ||
|
|
3f89e46d66 | ||
|
|
d02712ddc6 | ||
|
|
91f48c63ff | ||
|
|
cd16769e57 | ||
|
|
0a778e802f | ||
|
|
d4d5f4d1dd | ||
|
|
b1033ed7c6 | ||
|
|
253333a4cf | ||
|
|
50595ac3a9 | ||
|
|
37f1e55ed5 | ||
|
|
8ad14728f6 | ||
|
|
6b3cd3d31d | ||
|
|
fba698cb74 | ||
|
|
0946b123bb | ||
|
|
b43937a7a1 | ||
|
|
4564eba20d | ||
|
|
5cc0c0a3da | ||
|
|
f8e7c75f86 | ||
|
|
042cd354dc | ||
|
|
977bda97b2 | ||
|
|
4cf48ca9bb | ||
|
|
a247df50ea | ||
|
|
1f1c214944 | ||
|
|
d16db4ebde | ||
|
|
ae1c563082 | ||
|
|
ebd0edbab7 | ||
|
|
e87e9d269a | ||
|
|
187c64182b | ||
|
|
60c7ea6e5f | ||
|
|
6f1b4b0eee | ||
|
|
fb69300397 | ||
|
|
5677924525 | ||
|
|
8422fc632d | ||
|
|
d33cd744fb | ||
|
|
3b0fb27ae9 | ||
|
|
4603e09a04 | ||
|
|
7b0265ffe2 | ||
|
|
ec54560a38 | ||
|
|
6a5abd3672 | ||
|
|
3e6af39c42 | ||
|
|
cf82ffc052 | ||
|
|
b190150281 | ||
|
|
53ffe5df43 | ||
|
|
d39df8d3ed | ||
|
|
eeb2b928b9 | ||
|
|
6cf24a748f | ||
|
|
3b4fd180de | ||
|
|
7300c7a853 | ||
|
|
376f6db3ac | ||
|
|
ad3a960717 | ||
|
|
0a1ef867ee | ||
|
|
a7fe69deea | ||
|
|
a8c3b6d46f | ||
|
|
60db2655bb | ||
|
|
fd717b6995 | ||
|
|
ba37388fe3 | ||
|
|
68f32d85c9 | ||
|
|
2aa77e85de | ||
|
|
91665fdf0e | ||
|
|
23a1c64bf7 | ||
|
|
c2dcf06632 | ||
|
|
fad91bb818 | ||
|
|
0e769ece26 | ||
|
|
9a780b40a2 | ||
|
|
898873e9e3 | ||
|
|
52292e5f7e | ||
|
|
5de6c866b7 | ||
|
|
ec0cd3aec4 | ||
|
|
f5f9512d9a | ||
|
|
fb27cb4356 | ||
|
|
94664580c8 | ||
|
|
99a93fa9ea | ||
|
|
4cb6918506 | ||
|
|
9cc743bf84 |
No files matched your search
@@ -29,6 +29,19 @@ jobs:
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
@@ -51,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -153,6 +166,28 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
- name: Thunkgen tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
|
||||
|
||||
- name: Thunkgen Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
@@ -10,3 +10,4 @@ out/
|
||||
.vscode/
|
||||
.vs/
|
||||
*.pyc
|
||||
.cache
|
||||
@@ -49,3 +49,7 @@
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/Tessil/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
+36
-196
@@ -1,29 +1,34 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX)
|
||||
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with LLD" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_STATIC_PIE "Enables static-pie build" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
|
||||
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
|
||||
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
# These options are meant for package management
|
||||
@@ -41,6 +46,12 @@ if (ENABLE_ASSERTIONS)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GDB_SYMBOLS)
|
||||
message(STATUS "GDBSymbols support enabled")
|
||||
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
message(STATUS "Interpreter enabled")
|
||||
add_definitions(-DINTERPRETER_ENABLED=1)
|
||||
@@ -73,6 +84,7 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
@@ -99,9 +111,14 @@ if (ENABLE_COMPILE_TIME_TRACE)
|
||||
endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
if (ENABLE_LLD)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
link_libraries(${LD_OVERRIDE})
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
@@ -115,189 +132,13 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
endif()
|
||||
|
||||
# Check if the build target page size is 4096
|
||||
include(CheckCSourceRuns)
|
||||
|
||||
check_c_source_runs(
|
||||
"#include <unistd.h>
|
||||
int main(int argc, char* argv[])
|
||||
{
|
||||
return getpagesize() == 4096 ? 0 : 1;
|
||||
}"
|
||||
PAGEFILE_RESULT
|
||||
)
|
||||
|
||||
if (NOT ${PAGEFILE_RESULT})
|
||||
message(FATAL_ERROR "Host PAGE_SIZE is not 4096. Can't build on this target")
|
||||
endif()
|
||||
|
||||
include(CheckCXXSourceCompiles)
|
||||
check_cxx_source_compiles(
|
||||
"#include <sys/user.h>
|
||||
int main() {
|
||||
return PAGE_SIZE;
|
||||
}
|
||||
"
|
||||
HAS_PAGESIZE)
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"#include <sys/user.h>
|
||||
int main() {
|
||||
return PAGE_SHIFT;
|
||||
}
|
||||
"
|
||||
HAS_PAGESHIFT)
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"#include <sys/user.h>
|
||||
int main() {
|
||||
return PAGE_MASK;
|
||||
}
|
||||
"
|
||||
HAS_PAGEMASK)
|
||||
|
||||
if (NOT HAS_PAGESIZE)
|
||||
add_definitions(-DPAGE_SIZE=4096)
|
||||
endif()
|
||||
|
||||
if (NOT HAS_PAGESHIFT)
|
||||
add_definitions(-DPAGE_SHIFT=12)
|
||||
endif()
|
||||
if (NOT HAS_PAGEMASK)
|
||||
add_definitions("-DPAGE_MASK=(~(PAGE_SIZE-1))")
|
||||
endif()
|
||||
|
||||
if(DEFINED ENV{TERMUX_VERSION})
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_STATIC_PIE)
|
||||
if (_M_ARM_64 AND ENABLE_LLD)
|
||||
message (FATAL_ERROR "Static linking does not currently work with AArch64+LLD. Use GNU ld for now.")
|
||||
endif()
|
||||
|
||||
file(WRITE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
"int main(int argc, char* argv[])
|
||||
{
|
||||
return 0;
|
||||
}")
|
||||
|
||||
# Compile the test application with our LD_OVERRIDE and static-pie options
|
||||
try_compile(
|
||||
COMPILE_RESULT
|
||||
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp
|
||||
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
COMPILE_DEFINITIONS "-fPIE ${LD_OVERRIDE}"
|
||||
LINK_LIBRARIES "-static-pie ${LD_OVERRIDE}"
|
||||
COPY_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
|
||||
)
|
||||
|
||||
if (${COMPILE_RESULT})
|
||||
# Read the symbols from the elf
|
||||
execute_process(COMMAND
|
||||
readelf -s ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
|
||||
OUTPUT_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
|
||||
OUTPUT_VARIABLE PLT_SYMBOLS)
|
||||
|
||||
# Pull out the __rela_iplt_{start,end} symbols if they exist
|
||||
execute_process(COMMAND
|
||||
"grep" "__rela_iplt" ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
|
||||
OUTPUT_VARIABLE PLT_SYMBOLS)
|
||||
|
||||
set (SYMBOLS_FINE TRUE)
|
||||
set (HAS_IPLT -1)
|
||||
# Check if we have any symbols in our grep output
|
||||
# The symbols must either not exist at all OR the symbols are zero
|
||||
if (PLT_SYMBOLS)
|
||||
string(FIND ${PLT_SYMBOLS} "__rela_iplt_start" HAS_IPLT)
|
||||
endif()
|
||||
|
||||
if (NOT HAS_IPLT EQUAL -1)
|
||||
# We have some symbols from readelf. Let's parse the results to check if they are zero
|
||||
# Format: '35: 0000000000000000 0 NOTYPE LOCAL HIDDEN UND __rela_iplt_start'
|
||||
string(REPLACE "\n" ";" SYMBOL_LIST ${PLT_SYMBOLS})
|
||||
foreach (SYMBOL ${SYMBOL_LIST})
|
||||
# strip any leading and trailing whitespace
|
||||
string (STRIP ${SYMBOL} SYMBOL)
|
||||
# Convert string to a list
|
||||
string(REPLACE " " ";" SYMBOL_VALUES ${SYMBOL}})
|
||||
# Pull out the address argument
|
||||
list(GET SYMBOL_VALUES 1 OFFSET)
|
||||
|
||||
# Check against integer zero
|
||||
if (NOT ${OFFSET} EQUAL 0)
|
||||
# Symbol wasn't zero, this now fails
|
||||
set (SYMBOLS_FINE FALSE)
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if (SYMBOLS_FINE)
|
||||
# We can now exnable static-pie
|
||||
set (STATIC_PIE_OPTIONS "-static-pie")
|
||||
# Pthreads has an issue with exposing symbols
|
||||
# We need to make some concessions to the pthread gods
|
||||
if (ENABLE_LLD)
|
||||
set (PTHREAD_LIB
|
||||
-Wl,--undefined-glob=pthread_*
|
||||
-Wl,--undefined=__cxa_finalize
|
||||
-Wl,--undefined=_pthread_cleanup_push_defer
|
||||
-Wl,--undefined=_pthread_cleanup_pop_restore
|
||||
-Wl,--undefined=__pthread_cleanup_upto
|
||||
pthread)
|
||||
else()
|
||||
set (PTHREAD_LIB
|
||||
-Wl,--undefined=pthread_join
|
||||
-Wl,--undefined=pthread_attr_getdetachstate
|
||||
-Wl,--undefined=pthread_sigmask
|
||||
-Wl,--undefined=pthread_mutex_lock
|
||||
-Wl,--undefined=pthread_cond_init
|
||||
-Wl,--undefined=pthread_attr_init
|
||||
-Wl,--undefined=pthread_mutex_unlock
|
||||
-Wl,--undefined=pthread_mutexattr_destroy
|
||||
-Wl,--undefined=pthread_detach
|
||||
-Wl,--undefined=pthread_mutex_init
|
||||
-Wl,--undefined=pthread_getattr_np
|
||||
-Wl,--undefined=pthread_cond_timedwait
|
||||
-Wl,--undefined=pthread_attr_destroy
|
||||
-Wl,--undefined=pthread_mutexattr_settype
|
||||
-Wl,--undefined=pthread_rwlock_unlock
|
||||
-Wl,--undefined=pthread_rwlock_wrlock
|
||||
-Wl,--undefined=pthread_setspecific
|
||||
-Wl,--undefined=pthread_create
|
||||
-Wl,--undefined=pthread_cond_clockwait
|
||||
-Wl,--undefined=pthread_key_create
|
||||
-Wl,--undefined=pthread_rwlock_rdlock
|
||||
-Wl,--undefined=pthread_setname_np
|
||||
-Wl,--undefined=pthread_cond_signal
|
||||
-Wl,--undefined=pthread_mutexattr_init
|
||||
-Wl,--undefined=pthread_attr_setstack
|
||||
-Wl,--undefined=pthread_self
|
||||
-Wl,--undefined=pthread_getaffinity_np
|
||||
-Wl,--undefined=pthread_cond_wait
|
||||
-Wl,--undefined=pthread_mutex_trylock
|
||||
-Wl,--undefined=pthread_cond_broadcast
|
||||
-Wl,--undefined=pthread_cond_destroy
|
||||
-Wl,--undefined=pthread_getspecific
|
||||
-Wl,--undefined=pthread_key_delete
|
||||
-Wl,--undefined=pthread_once
|
||||
-Wl,--undefined=__cxa_finalize
|
||||
-Wl,--undefined=_pthread_cleanup_push_defer
|
||||
-Wl,--undefined=_pthread_cleanup_pop_restore
|
||||
-Wl,--undefined=__pthread_cleanup_upto
|
||||
pthread)
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Application has __rela_iplt_{start,end} symbols. Which means static-pie can't be enabled")
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Couldn't compile static-pie test. Static-pie can't be enabled! Is your glibc compiled without static-pie?")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
add_definitions(-DENABLE_ASAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
@@ -533,15 +374,20 @@ add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
# Install the ThunksDB file
|
||||
install(
|
||||
FILES ${CMAKE_CURRENT_SOURCE_DIR}/Data/ThunksDB.json
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
# Thunk targets for both host libraries and IDE integration
|
||||
@@ -558,10 +404,10 @@ if (BUILD_THUNKS)
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DX86_C_COMPILER:STRING=${X86_C_COMPILER}"
|
||||
"-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
@@ -628,15 +474,9 @@ endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
if (ENABLE_STATIC_PIE)
|
||||
set (CPACK_PACKAGE_NAME fex-emu-static)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu")
|
||||
else()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu-static")
|
||||
endif()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.org>")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
|
||||
+1
-1
@@ -55,7 +55,7 @@ further defined and clarified by project maintainers.
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported by contacting the project team at team@fex-emu.org. All
|
||||
reported by contacting the project team at team@fex-emu.com. All
|
||||
complaints will be reviewed and investigated and will result in a response that
|
||||
is deemed necessary and appropriate to the circumstances. The project team is
|
||||
obligated to maintain confidentiality with regard to the reporter of an incident.
|
||||
|
||||
@@ -11,15 +11,15 @@ endforeach()
|
||||
# First generate then install it
|
||||
foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Get the filename only component
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WE)
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
|
||||
|
||||
# Configure it
|
||||
configure_file(
|
||||
${GEN_CONFIG_SRC}
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json)
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
|
||||
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
endforeach()
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"AdditionalArguments": "--no-sandbox"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
+60
-239
@@ -6,18 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -26,322 +18,151 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan-radeon": {
|
||||
"Library": "libvulkan_radeon-guest.so",
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_radeon.so"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
|
||||
]
|
||||
},
|
||||
"Vulkan-lavapipe": {
|
||||
"Library": "libvulkan_lvp-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_lvp.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-freedreno": {
|
||||
"Library": "libvulkan_freedreno-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_freedreno.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-intel": {
|
||||
"Library": "libvulkan_intel-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_intel.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-panfrost": {
|
||||
"Library": "libvulkan_panfrost-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_panfrost.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-nvidia": {
|
||||
"Library": "libvulkan_nvidia-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/lib/x86_64-linux-gnu/libGLX_nvidia.so.0"
|
||||
],
|
||||
"Comment": [
|
||||
"Not currently wired up"
|
||||
]
|
||||
},
|
||||
"Vulkan-virtio": {
|
||||
"Library": "libvulkan_virtio-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_virtio.so"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb_dri2-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb_dri3-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb_xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb_shm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb_sync-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb_randr-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb_present-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb_glx-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
Vendored
+2
-2
@@ -46,14 +46,14 @@ if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
|
||||
Vendored
+2
-12
@@ -18,14 +18,8 @@ This project aims to provide a fast and functional x86-64 emulation library that
|
||||
* Portable library implementation in order to support easy integration in to applications
|
||||
### Target Host Architecture
|
||||
The target host architecture for this library is AArch64. Specifically the ARMv8.1 version or newer.
|
||||
The CPU IR is designed with AArch64 in mind but there is a desire to run the recompiled code on other architectures as well.
|
||||
Multiple architecture support is desired for easier bringup and debugging, performance isn't as much of a priority there (ex. x86-64(guest) translated to x86-64(host))
|
||||
### Not currently goals but will be in the future
|
||||
* 32bit x86 support
|
||||
* This will be a desire in the future, but to lower the amount of work required, decided to push this off for now.
|
||||
* Integration in to WINE
|
||||
* Later generation of x86-64 instruction sets
|
||||
* Including AVX, F16C, XOP, FMA, AVX2, etc
|
||||
The CPU IR is designed with AArch64 in mind but should allow for other architectures as well.
|
||||
x86-64 host support is available for ease of development, but is not a priority.
|
||||
### Not desired
|
||||
* Kernel space emulation
|
||||
* CPL0-2 emulation
|
||||
@@ -33,7 +27,3 @@ Multiple architecture support is desired for easier bringup and debugging, perfo
|
||||
* IRQs
|
||||
* SVM
|
||||
* "Cycle Accurate" emulation
|
||||
### Dependencies
|
||||
* clang-tidy if you want to ensure the code stays tidy
|
||||
* cmake
|
||||
* A C++17 compliant compiler (There are assumptions made about using Clang and LTO)
|
||||
-10
@@ -321,8 +321,6 @@ def print_ir_structs(defines):
|
||||
|
||||
|
||||
if op.SSAArgNum > 0:
|
||||
# Add helpers for accessing SSA arguments, given how frequently they're accessed
|
||||
|
||||
output_file.write("\t// Get index of argument by name\n")
|
||||
SSAArg = 0
|
||||
for arg in op.Arguments:
|
||||
@@ -330,14 +328,6 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tstatic constexpr size_t {}_Index = {};\n".format(arg.Name, SSAArg))
|
||||
SSAArg = SSAArg + 1
|
||||
|
||||
output_file.write("\n")
|
||||
output_file.write("\t[[nodiscard]] OrderedNodeWrapper& Args(size_t Index) {\n")
|
||||
output_file.write("\t\treturn Header.Args[Index];\n")
|
||||
output_file.write("\t}\n")
|
||||
output_file.write("\t[[nodiscard]] const OrderedNodeWrapper& Args(size_t Index) const {\n")
|
||||
output_file.write("\t\treturn Header.Args[Index];\n")
|
||||
output_file.write("\t}\n")
|
||||
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
|
||||
+15
-10
@@ -80,16 +80,20 @@ set (SRCS
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/CompileService.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
Interface/Core/OpcodeDispatcher/Vector.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
@@ -114,6 +118,7 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
@@ -184,7 +189,9 @@ if (ENABLE_JIT_X86_64)
|
||||
Interface/Core/JIT/x86_64/MemoryOps.cpp
|
||||
Interface/Core/JIT/x86_64/MiscOps.cpp
|
||||
Interface/Core/JIT/x86_64/MoveOps.cpp
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp)
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp
|
||||
Interface/Core/JIT/x86_64/x64Relocations.cpp
|
||||
)
|
||||
list(APPEND DEFINES -DJIT_X86_64)
|
||||
endif()
|
||||
|
||||
@@ -201,7 +208,9 @@ if (ENABLE_JIT_ARM64)
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
@@ -219,13 +228,11 @@ set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
add_custom_target(CREATE_IR_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_IR_FOLDER}")
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
)
|
||||
@@ -239,7 +246,6 @@ set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_IR_DOC}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
|
||||
)
|
||||
@@ -260,15 +266,13 @@ set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
|
||||
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
|
||||
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
|
||||
|
||||
add_custom_target(CREATE_CONFIG_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
|
||||
file(MAKE_DIRECTORY "${OUTPUT_CONFIG_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_CONFIG_NAME}"
|
||||
OUTPUT "${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
OUTPUT "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${INPUT_CONFIG_NAME}"
|
||||
DEPENDS CREATE_CONFIG_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
@@ -329,6 +333,7 @@ function(AddDefaultOptionsToTarget Name)
|
||||
|
||||
-Wno-trigraphs
|
||||
-ffunction-sections
|
||||
-fwrapv
|
||||
)
|
||||
|
||||
if (GCC_COLOR)
|
||||
|
||||
+3
-3
@@ -20,12 +20,12 @@ struct BitSet final {
|
||||
ElementType *Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
@@ -64,7 +64,7 @@ struct BitSetView final {
|
||||
ElementType *Memory;
|
||||
|
||||
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0,
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
+13
-2
@@ -7,6 +7,11 @@
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fp.reset(fopen(PerfMap.c_str(), "wb"));
|
||||
@@ -16,8 +21,6 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
@@ -34,6 +37,14 @@ namespace FEXCore {
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (!fp) return;
|
||||
|
||||
|
||||
+2
@@ -11,8 +11,10 @@ public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
|
||||
|
||||
+5
-1
@@ -188,6 +188,10 @@ struct X80SoftFloat {
|
||||
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
|
||||
return extF80_roundToInt(lhs, RoundMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
@@ -257,7 +261,7 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs);
|
||||
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
|
||||
BIGFLOAT Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore::StringUtils {
|
||||
// Trim the left side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
// Trim the right side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
// Trim both the left and right of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
|
||||
}
|
||||
}
|
||||
+48
-34
@@ -1,4 +1,5 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Common/Paths.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
@@ -81,7 +82,7 @@ namespace JSON {
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
LogMan::Msg::EFmt("Couldn't get config list");
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -153,15 +154,20 @@ namespace JSON {
|
||||
return ConfigDir;
|
||||
}
|
||||
|
||||
std::string GetConfigFileLocation() {
|
||||
std::string GetConfigFileLocation(bool Global) {
|
||||
std::string ConfigFile{};
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
if (Global) {
|
||||
ConfigFile = GetConfigDirectory(true) + "Config.json";
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
}
|
||||
}
|
||||
return ConfigFile;
|
||||
}
|
||||
@@ -205,7 +211,8 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 6> LoadOrder = {
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
@@ -370,29 +377,22 @@ namespace JSON {
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string ltrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
|
||||
std::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
}
|
||||
|
||||
return String;
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string rtrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string trim(std::string String) {
|
||||
return rtrim(ltrim(String));
|
||||
}
|
||||
|
||||
|
||||
std::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
@@ -401,7 +401,7 @@ namespace JSON {
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = trim(ManagerStr);
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
@@ -445,6 +445,16 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
// If running the interpreter then disable cache code compilation
|
||||
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
@@ -609,7 +619,7 @@ namespace JSON {
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader();
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
@@ -655,9 +665,9 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader()
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation()} {
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
@@ -731,12 +741,16 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>();
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+85
-15
@@ -38,6 +38,24 @@
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
"TextDefault": "none",
|
||||
"Choices": [ "none", "read", "readwrite" ],
|
||||
"ArgumentHandler": "CacheObjectCodeHandler",
|
||||
"Desc": [
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
},
|
||||
"EnableAVX": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Determines whether or not we use the expanded register file for AVX or not"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
@@ -104,6 +122,13 @@
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
|
||||
]
|
||||
},
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
@@ -190,6 +215,16 @@
|
||||
"Useful for determining hot blocks of code",
|
||||
"Has some file writing overhead per JIT block"
|
||||
]
|
||||
},
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
"Also needs x86_64-linux-gnu-objdump in PATH.",
|
||||
"Can be very slow."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
@@ -201,36 +236,28 @@
|
||||
"Disables logging"
|
||||
]
|
||||
},
|
||||
"OutputSocket": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Socket to connect to",
|
||||
"eg: localhost:8087",
|
||||
"If set will override the OutputLog location"
|
||||
]
|
||||
},
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "stderr",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, <Filename>]"
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MMAN",
|
||||
"TextDefault": "mman",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
"Checks code for modification before execution.",
|
||||
"\tnone: No checks",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap",
|
||||
"\tfull: Validate code before every run (slow)"
|
||||
"\tmtrack: Page tracking based invalidation",
|
||||
"\tfull: Validate code before every run (slow)",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
|
||||
]
|
||||
},
|
||||
"TSOEnabled": {
|
||||
@@ -241,6 +268,21 @@
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -274,6 +316,16 @@
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
]
|
||||
},
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
|
||||
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
|
||||
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
|
||||
"Can be useful for Wine applications that rely on stack unwinding"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
@@ -299,6 +351,13 @@
|
||||
"Desc": [
|
||||
"Loads an AOT IR cache for the loaded executable."
|
||||
]
|
||||
},
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -316,6 +375,17 @@
|
||||
"Type": "str",
|
||||
"Default": ""
|
||||
},
|
||||
"APP_CONFIG_NAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"This is the application config name that has been loaded.",
|
||||
"This differs from APP_FILENAME in two ways",
|
||||
"Where APP_FILENAME always points to the executable path that FEX-Emu is executing.",
|
||||
"This matches what is used to load the AppLayer configuration name.",
|
||||
"When running through a compatibility layer like wine, this will only be the exe name, instead of wine full path."
|
||||
]
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
|
||||
+20
-7
@@ -43,8 +43,8 @@ namespace FEXCore::Context {
|
||||
delete CTX;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
|
||||
return CTX->InitCore(Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
return CTX->InitCore(InitialRIP, StackPointer);
|
||||
}
|
||||
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
|
||||
@@ -110,6 +110,10 @@ namespace FEXCore::Context {
|
||||
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
|
||||
}
|
||||
|
||||
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->HostFeatures;
|
||||
}
|
||||
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
CTX->HandleCallback(Thread, RIP);
|
||||
}
|
||||
@@ -149,13 +153,14 @@ namespace FEXCore::Context {
|
||||
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->CleanupAfterFork(Thread);
|
||||
}
|
||||
|
||||
|
||||
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
|
||||
CTX->SignalDelegation = SignalDelegation;
|
||||
}
|
||||
|
||||
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
|
||||
CTX->SyscallHandler = Handler;
|
||||
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
|
||||
@@ -186,11 +191,19 @@ namespace FEXCore::Context {
|
||||
CTX->WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
|
||||
return CTX->AddNamedRegion(Base, Length, Offset, Name);
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
|
||||
return CTX->LoadAOTIRCacheEntry(Name);
|
||||
}
|
||||
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
|
||||
return CTX->RemoveNamedRegion(Base, Length);
|
||||
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
|
||||
return CTX->UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(FEXCore::Context::Context *CTX, std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
CTX->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
|
||||
+83
-35
@@ -1,13 +1,16 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
@@ -34,13 +37,21 @@ class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
}
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
class InterpreterCore;
|
||||
class Dispatcher;
|
||||
}
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -67,6 +78,7 @@ namespace FEXCore::Context {
|
||||
friend class FEXCore::CPU::X86JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::CPU::InterpreterCore;
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
@@ -81,6 +93,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
@@ -97,19 +110,21 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
IntCallbackReturn InterpreterCallbackReturn;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
uint64_t ThreadID{};
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
std::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
std::mutex IdleWaitMutex;
|
||||
std::condition_variable IdleWaitCV;
|
||||
@@ -118,9 +133,13 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
@@ -135,7 +154,7 @@ namespace FEXCore::Context {
|
||||
Context();
|
||||
~Context();
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus() const;
|
||||
bool IsPaused() const { return !Running; }
|
||||
@@ -155,13 +174,33 @@ namespace FEXCore::Context {
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState
|
||||
static void RemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
RemoveCodeEntry(Frame->Thread, GuestRIP);
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
uint64_t GetThreadCount() const;
|
||||
@@ -171,21 +210,19 @@ namespace FEXCore::Context {
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
@@ -196,21 +233,9 @@ namespace FEXCore::Context {
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
* @param CompileThread Is this for the compile service or not?
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
* This is exposed because the CompileService needs to initialize compilers while copying data from
|
||||
* the paired InternalThreadState that it is compiling code for
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
@@ -233,7 +258,7 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
|
||||
/**
|
||||
* @brief Initializes the TLS data for a thread
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
@@ -269,8 +294,8 @@ namespace FEXCore::Context {
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
|
||||
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
@@ -297,8 +322,17 @@ namespace FEXCore::Context {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions);
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
|
||||
void MarkMemoryShared();
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
private:
|
||||
/**
|
||||
@@ -310,22 +344,36 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
|
||||
// Entry Cache
|
||||
uint64_t StartingRIP;
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
|
||||
|
||||
+166
-115
@@ -1,14 +1,15 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <signal.h>
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include <csignal>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
@@ -513,7 +514,8 @@ uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
//Only 32-bit pairs
|
||||
for(int i = 1; i < 10; i++) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg1 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::CCMP_MASK) == FEXCore::ArchHelpers::Arm64::CCMP_INST) {
|
||||
ExpectedReg2 = GetRmReg(NextInstr);
|
||||
@@ -1579,7 +1581,7 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
@@ -1592,7 +1594,7 @@ bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
if (Size == 2) {
|
||||
auto Res = DoLoad16(Addr);
|
||||
@@ -1622,7 +1624,7 @@ bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
@@ -1635,7 +1637,7 @@ bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
constexpr bool DoRetry = false;
|
||||
if (Size == 2) {
|
||||
@@ -1745,7 +1747,8 @@ static uint64_t HandleCAS_NoAtomics(void *_ucontext, void *_info)
|
||||
#endif
|
||||
DesiredReg = GetRdReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg = GetRmReg(NextInstr);
|
||||
}
|
||||
}
|
||||
@@ -1828,11 +1831,13 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
// Scan forward at most five instructions to find our instructions
|
||||
for (size_t i = 1; i < 6; ++i) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_SHIFT_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ADD;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_SHIFT_INST) {
|
||||
uint32_t RnReg = GetRnReg(NextInstr);
|
||||
if (RnReg == REGISTER_MASK) {
|
||||
// Zero reg means neg
|
||||
@@ -1843,21 +1848,34 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST ) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::AND_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_AND;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::BIC_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_BIC;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::OR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_OR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ORN_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ORN;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EOR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EOR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EON_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EON;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
@@ -1888,40 +1906,53 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
constexpr bool DoRetry = true;
|
||||
|
||||
auto NOPExpected = []<typename AtomicType>(AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto BICDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & ~Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto ORNDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | ~Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto EONDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ ~Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
if (Size == 2) {
|
||||
using AtomicType = uint16_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -1937,17 +1968,27 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1966,38 +2007,6 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
else if (Size == 4) {
|
||||
using AtomicType = uint32_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -2013,17 +2022,27 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2042,38 +2061,6 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
else if (Size == 8) {
|
||||
using AtomicType = uint64_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -2089,17 +2076,27 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2140,7 +2137,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
@@ -2164,7 +2161,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
@@ -2186,6 +2183,60 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t LDUR = 0b0011'1000'0100'0000'0000'0000'0000'0000;
|
||||
LDUR |= Size << 30;
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t STUR = 0b0011'1000'0000'0000'0000'0000'0000'0000;
|
||||
STUR |= Size << 30;
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXP_MASK) == FEXCore::ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
//Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleCASPAL_ARMv8(ucontext, info, Instr);
|
||||
|
||||
+22
-9
@@ -12,6 +12,10 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
|
||||
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
|
||||
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
|
||||
|
||||
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
|
||||
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
|
||||
|
||||
@@ -27,13 +31,19 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
|
||||
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
|
||||
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t EON_INST = 0x4A'20'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
@@ -46,8 +56,11 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
TYPE_ADD,
|
||||
TYPE_SUB,
|
||||
TYPE_AND,
|
||||
TYPE_BIC,
|
||||
TYPE_OR,
|
||||
TYPE_ORN,
|
||||
TYPE_EOR,
|
||||
TYPE_EON,
|
||||
TYPE_NEG, // This is just a sub with zero. Need to know the differences
|
||||
};
|
||||
|
||||
@@ -83,8 +96,8 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
|
||||
+104
-46
@@ -1,22 +1,28 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "cpu-features.h"
|
||||
#include "aarch64/instructions-aarch64.h"
|
||||
#include "utils-vixl.h"
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
#include <cpu-features.h>
|
||||
#include <utils-vixl.h>
|
||||
|
||||
#include <array>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
@@ -42,12 +48,57 @@ void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant,
|
||||
}
|
||||
|
||||
int NumMoves = 1;
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
// We only want to use ADRP+ADD if we have more than 1 segment
|
||||
for (size_t i = 0; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
if (Part != 0) {
|
||||
++RequiredMoveSegments;
|
||||
}
|
||||
}
|
||||
|
||||
// ADRP+ADD is specifically optimized in hardware
|
||||
// Check if we can use this
|
||||
auto PC = GetCursorAddress<uint64_t>();
|
||||
|
||||
// PC aligned to page
|
||||
uint64_t AlignedPC = PC & ~0xFFFULL;
|
||||
|
||||
// Offset from aligned PC
|
||||
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
}
|
||||
else {
|
||||
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
|
||||
// 21-bit signed integer here
|
||||
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
|
||||
if (vixl::IsInt21(SmallOffset)) {
|
||||
adr(Reg, SmallOffset);
|
||||
}
|
||||
else {
|
||||
// Need to use ADRP + ADD
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
add(Reg, Reg, Constant & 0xFFF);
|
||||
NumMoves = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -158,19 +209,29 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
auto Reg1 = SRAFPR[i];
|
||||
auto Reg2 = SRAFPR[i+1];
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i+1][0])));
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -195,19 +256,29 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
auto Reg1 = SRAFPR[i];
|
||||
auto Reg2 = SRAFPR[i+1];
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i+1][0])));
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -260,23 +331,10 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
add(sp, sp, SPOffset);
|
||||
}
|
||||
|
||||
void Arm64Emitter::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,15 +1,19 @@
|
||||
#pragma once
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "platform-vixl.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -60,6 +64,7 @@ class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
@@ -77,11 +82,8 @@ protected:
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
|
||||
void ResetStack();
|
||||
void Align16B();
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
|
||||
@@ -124,7 +124,7 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
@@ -143,7 +143,7 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Host FPR state starts at _mcontext->reserved[0];
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
Backup->FPSR = HostState->FPSR;
|
||||
Backup->FPCR = HostState->FPCR;
|
||||
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
@@ -163,7 +163,7 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
HostState->FPCR = Backup->FPCR;
|
||||
HostState->FPSR = Backup->FPSR;
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = &CodeBuffers[0];
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (ThreadState->CTX->Config.GlobalJITNaming()) {
|
||||
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
for (auto &Buffer: CodeBuffers) {
|
||||
auto start = (uintptr_t)Buffer.Ptr;
|
||||
auto end = start + Buffer.Size;
|
||||
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
+31
-6
@@ -8,11 +8,11 @@ $end_info$
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
@@ -39,6 +39,7 @@ namespace ProductNames {
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
@@ -83,7 +84,10 @@ static uint32_t CalculateNumberOfCPUs() {
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
@@ -151,7 +155,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 35> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 36> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
@@ -161,6 +165,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
@@ -416,7 +421,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(0 << 1) | // PCLMULQDQ
|
||||
(1 << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(0 << 4) | // DS-CPL
|
||||
@@ -446,7 +451,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(0 << 31); // Hypervisor always returns zero
|
||||
(1 << 31); // Hypervisor always returns one
|
||||
|
||||
Res.edx =
|
||||
(1 << 0) | // FPU
|
||||
@@ -658,7 +663,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // Reserved
|
||||
(0 << 29) | // SHA instructions
|
||||
(1 << 29) | // SHA instructions
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
@@ -826,7 +831,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) {
|
||||
// CPUID documentation information:
|
||||
// 4000_0000h - 4FFF_FFFFh - No existing or future CPU will return information in this range
|
||||
// Reserved entirely for VMs to do whatever they want.
|
||||
Res.eax = 0x40000000;
|
||||
Res.eax = 0x40000001;
|
||||
|
||||
// EBX, EDX, ECX become the hypervisor ID signature
|
||||
constexpr static char HypervisorID[12] = "FEXIFEXIEMU";
|
||||
@@ -834,6 +839,25 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Leaf == 0) {
|
||||
// EAX[3:0] Is the host architecture that FEX is running under
|
||||
#ifdef _M_X86_64
|
||||
// EAX[3:0] = 1 = x86_64 host architecture
|
||||
Res.eax |= 0b0001;
|
||||
#elif defined(_M_ARM_64)
|
||||
// EAX[3:0] = 2 = AArch64 host architecture
|
||||
Res.eax |= 0b0010;
|
||||
#else
|
||||
// EAX[3:0] = 0 = Unknown architecture
|
||||
#endif
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Highest extended function implemented
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
@@ -1228,6 +1252,7 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
#endif
|
||||
// Hypervisor CPUID information leaf
|
||||
RegisterFunction(0x4000'0000, &CPUIDEmu::Function_4000_0000h);
|
||||
RegisterFunction(0x4000'0001, &CPUIDEmu::Function_4000_0001h);
|
||||
|
||||
// Largest extended function number
|
||||
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
|
||||
|
||||
@@ -80,6 +80,7 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
|
||||
@@ -1,165 +0,0 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
#include "FEXCore/HLE/Linux/ThreadManagement.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <memory>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
|
||||
namespace FEXCore {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CompileService *This = reinterpret_cast<FEXCore::CompileService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
CompileService::CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ParentThread {Thread} {
|
||||
|
||||
CompileThreadData = std::make_unique<FEXCore::Core::InternalThreadState>();
|
||||
CompileThreadData->IsCompileService = true;
|
||||
|
||||
// We need a compiler for this work thread
|
||||
CTX->InitializeCompiler(CompileThreadData.get(), true);
|
||||
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CompileService::Initialize() {
|
||||
// Share CompileService which = this
|
||||
CompileThreadData->CompileService = ParentThread->CompileService;
|
||||
}
|
||||
|
||||
void CompileService::Shutdown() {
|
||||
ShuttingDown = true;
|
||||
// Kick the working thread
|
||||
StartWork.NotifyAll();
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
|
||||
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
|
||||
if (CompileMutex.try_lock()) {
|
||||
// We can only clear these things if we pulled the compile mutex
|
||||
|
||||
// Grab the work queue and clear it
|
||||
// We don't need to grab the queue mutex since this thread will no longer receive any work events
|
||||
// Threads are bounded 1:1
|
||||
while (!WorkQueue.empty()) {
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Go through the garbage collection array and clear it
|
||||
// It's safe to clear things that aren't marked safe since we are clearing cache
|
||||
GCArray.clear();
|
||||
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LocalIRCache.empty(), "Compile service must never have LocalIRCache");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
|
||||
// Clear the inverse cache of what is calling us from the Context ClearCache routine
|
||||
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
|
||||
SelectedThread->LookupCache->ClearCache();
|
||||
SelectedThread->CPUBackend->ClearCache();
|
||||
}
|
||||
|
||||
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
|
||||
WorkItem* ResultItem = nullptr;
|
||||
|
||||
{
|
||||
// Tell the worker thread to compile code for us
|
||||
auto Item = std::make_unique<WorkItem>();
|
||||
Item->RIP = RIP;
|
||||
|
||||
// Fill the threads work queue
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
ResultItem = WorkQueue.emplace(std::move(Item)).get();
|
||||
}
|
||||
|
||||
// Notify the thread that it has more work
|
||||
StartWork.NotifyAll();
|
||||
|
||||
return ResultItem;
|
||||
}
|
||||
|
||||
void CompileService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16]{};
|
||||
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
|
||||
while (true) {
|
||||
// Wait for work
|
||||
StartWork.Wait();
|
||||
if (ShuttingDown.load()) {
|
||||
break;
|
||||
}
|
||||
|
||||
std::scoped_lock lk(CompileMutex);
|
||||
size_t WorkItems{};
|
||||
|
||||
do {
|
||||
// Grab a work item
|
||||
std::unique_ptr<WorkItem> Item{};
|
||||
{
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
Item = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
}
|
||||
|
||||
// If we had a work item then work on it
|
||||
if (Item) {
|
||||
// Make sure it's not in lookup cache by accident
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->CurrentFrame->State.rip = Item->RIP;
|
||||
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
if (!CodePtr) {
|
||||
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
|
||||
ERROR_AND_DIE_FMT("Couldn't compile code for thread at RIP: 0x{:x}", Item->RIP);
|
||||
}
|
||||
|
||||
Item->CodePtr = CodePtr;
|
||||
Item->IRList = IRList;
|
||||
Item->DebugData = DebugData;
|
||||
Item->RAData = RAData;
|
||||
Item->StartAddr = StartAddr;
|
||||
Item->Length = Length;
|
||||
|
||||
auto& GCItem = GCArray.emplace_back(std::move(Item));
|
||||
GCItem->ServiceWorkDone.NotifyAll();
|
||||
}
|
||||
} while (WorkItems != 0);
|
||||
|
||||
// Clean up any safe entries in our GC array if we have any.
|
||||
std::erase_if(GCArray, [](const auto& Entry) {
|
||||
return Entry->SafeToClear.load(std::memory_order_relaxed);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,68 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <stdint.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
};
|
||||
class CompileService final {
|
||||
public:
|
||||
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
void Initialize();
|
||||
void Shutdown();
|
||||
|
||||
struct WorkItem {
|
||||
// Incoming
|
||||
uint64_t RIP{};
|
||||
|
||||
// Outgoing
|
||||
void *CodePtr{};
|
||||
FEXCore::IR::IRListView *IRList{};
|
||||
FEXCore::IR::RegisterAllocationData *RAData{};
|
||||
FEXCore::Core::DebugData *DebugData{};
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
|
||||
// Communication
|
||||
Event ServiceWorkDone{};
|
||||
std::atomic_bool SafeToClear{};
|
||||
};
|
||||
|
||||
WorkItem *CompileCode(uint64_t RIP);
|
||||
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address) const {
|
||||
return CompileThreadData->CPUBackend->IsAddressInJITCode(Address, false, false);
|
||||
}
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ParentThread;
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::unique_ptr<FEXCore::Core::InternalThreadState> CompileThreadData;
|
||||
|
||||
std::mutex QueueMutex{};
|
||||
std::mutex CompileMutex{};
|
||||
std::queue<std::unique_ptr<WorkItem>> WorkQueue{};
|
||||
std::vector<std::unique_ptr<WorkItem>> GCArray{};
|
||||
Event StartWork{};
|
||||
std::atomic_bool ShuttingDown{false};
|
||||
};
|
||||
}
|
||||
+482
-273
File diff suppressed because it is too large.
Load diff
+232
-116
@@ -16,20 +16,23 @@
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "code-buffer-vixl.h"
|
||||
#include "platform-vixl.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <code-buffer-vixl.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
@@ -38,12 +41,11 @@ using namespace vixl::aarch64;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SetAllowAssembler(true);
|
||||
|
||||
DispatchPtr = GetCursorAddress<CPUBackend::AsmDispatch>();
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
|
||||
// while (true) {
|
||||
// Ptr = FindBlock(RIP)
|
||||
@@ -53,12 +55,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// Ptr();
|
||||
// }
|
||||
|
||||
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
Literal l_CompileBlock {GetCompileBlockPtr()};
|
||||
Literal l_ExitFunctionLink {config.ExitFunctionLink};
|
||||
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -71,11 +70,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
add(x0, sp, 0);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
|
||||
@@ -92,11 +91,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify x2 since it contains our RIP once the block doesn't exist
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
|
||||
auto RipReg = x2;
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -104,21 +103,17 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
br(x3);
|
||||
} else {
|
||||
b(&CallBlock);
|
||||
}
|
||||
br(x3);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(x0, &l_PagePtr);
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, VirtualMemorySize - 1);
|
||||
}
|
||||
@@ -157,44 +152,21 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
|
||||
// Jump to the block
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
br(x3);
|
||||
} else {
|
||||
bind(&CallBlock);
|
||||
mov(x0, STATE);
|
||||
blr(x3);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, &l_CTX);
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
// If the value == 0 then branch to the top
|
||||
cbz(x0, &LoopTop);
|
||||
// Else we need to pause now
|
||||
b(&ThreadPauseHandler);
|
||||
} else {
|
||||
// Unconditionally loop to the top
|
||||
// We will only stop on error when compiling a block or signal
|
||||
b(&LoopTop);
|
||||
}
|
||||
}
|
||||
br(x3);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
@@ -209,7 +181,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -231,11 +203,10 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
svc(0);
|
||||
}
|
||||
|
||||
ldr(x0, &l_ExitFunctionLinkThis);
|
||||
mov(x1, STATE);
|
||||
mov(x2, lr);
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
|
||||
ldr(x3, &l_ExitFunctionLink);
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
blr(x3);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -256,7 +227,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
mov(x0, x4);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
br(x0);
|
||||
}
|
||||
@@ -265,7 +236,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
bind(&NoBlock);
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -312,7 +283,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
add(sp, sp, 16);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
b(&LoopTop);
|
||||
@@ -329,42 +300,44 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, X86State::X86_TRAPNO_OF);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)));
|
||||
LoadConstant(w1, 0x80);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)));
|
||||
LoadConstant(x1, 0);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)));
|
||||
brk(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(x1, 0);
|
||||
ldr(x1, MemOperand(x1));
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
bind(&ThreadPauseHandler);
|
||||
@@ -399,7 +372,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
|
||||
// When the thunk itself returns, it'll do its regular return logic there
|
||||
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
CallbackPtr = GetCursorAddress<CPUBackend::JITCallback>();
|
||||
CallbackPtr = GetCursorAddress<JITCallback>();
|
||||
|
||||
// We expect the thunk to have previously pushed the registers it was using
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -408,40 +381,112 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
mov(STATE, x0);
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
|
||||
ldr(w2, MemOperand(x0));
|
||||
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
add(w2, w2, 1);
|
||||
str(w2, MemOperand(x0));
|
||||
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(x2, x2, 16);
|
||||
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
str(x0, MemOperand(x2));
|
||||
|
||||
// Store RIP to the context state
|
||||
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
str(x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// load static regs
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
place(&l_PagePtr);
|
||||
{
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
place(&l_CTX);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
place(&l_ExitFunctionLink);
|
||||
place(&l_ExitFunctionLinkThis);
|
||||
|
||||
|
||||
FinalizeCode();
|
||||
@@ -457,51 +502,122 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
}
|
||||
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
Pointers.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Pointers.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Pointers.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Pointers.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Pointers.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
|
||||
Pointers.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
|
||||
Pointers.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Pointers.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
|
||||
for(int i = 0; i < SRA64.size(); i++) {
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
|
||||
emit.ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
emit.ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(w0, &RunBlock);
|
||||
{
|
||||
Literal l_GuestRIP {GuestRIP};
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(x0, &l_GuestRIP);
|
||||
emit.str(x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(x0);
|
||||
emit.place(&l_GuestRIP);
|
||||
}
|
||||
emit.bind(&RunBlock);
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label InlineIRData;
|
||||
|
||||
emit.mov(x0, STATE);
|
||||
emit.adr(x1, &InlineIRData);
|
||||
|
||||
emit.ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
emit.blr(x3);
|
||||
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
emit.br(x0);
|
||||
|
||||
emit.bind(&InlineIRData);
|
||||
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
}
|
||||
|
||||
for(int i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&ThreadState->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
AArch64.LDIVHandler = LDIVHandlerAddress;
|
||||
AArch64.LUREMHandler = LUREMHandlerAddress;
|
||||
AArch64.LREMHandler = LREMHandlerAddress;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -15,10 +15,20 @@ namespace FEXCore::CPU {
|
||||
|
||||
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
public:
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
protected:
|
||||
void SpillSRA(void *ucontext, uint32_t IgnoreMask) override;
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress{};
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
};
|
||||
|
||||
}
|
||||
+149
-95
@@ -1,8 +1,8 @@
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -40,7 +40,7 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
|
||||
// We can end up getting a signal at any point in our host state
|
||||
// Jump to a handler that saves all state so we can safely return
|
||||
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -65,7 +65,7 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
// Save guest state
|
||||
// We can't guarantee if registers are in context or host GPRs
|
||||
// So we need to save everything
|
||||
memcpy(&Context->GuestState, ThreadState->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Set the new SP
|
||||
ArchHelpers::Context::SetSp(ucontext, NewSP);
|
||||
@@ -82,13 +82,13 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
}
|
||||
|
||||
void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
|
||||
uint64_t OldSP{};
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
|
||||
OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -99,17 +99,18 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
SignalFrames.pop();
|
||||
}
|
||||
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
uintptr_t NewSP = OldSP;
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
|
||||
// First thing, reset the guest state
|
||||
memcpy(ThreadState->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Now restore host state
|
||||
ArchHelpers::Context::RestoreContext(ucontext, Context);
|
||||
|
||||
if (Context->UContextLocation) {
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
|
||||
// XXX: Unsupported since it needs state reconstruction
|
||||
@@ -133,7 +134,7 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
@@ -159,10 +160,22 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
memcpy(Frame->State.xmm, fpstate->_xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
@@ -191,7 +204,7 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
@@ -216,16 +229,26 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
@@ -274,14 +297,34 @@ static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Signal, ucontext);
|
||||
template <typename T>
|
||||
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
fpstate->sw_reserved.magic1 = x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
|
||||
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
|
||||
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_FP |
|
||||
x86_64::fpx_sw_bytes::FEATURE_SSE;
|
||||
if (is_avx_enabled) {
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_YMM;
|
||||
}
|
||||
|
||||
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
|
||||
|
||||
if (is_avx_enabled) {
|
||||
xstate->xstate_hdr.xfeatures = 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
// Set the new PC
|
||||
@@ -293,14 +336,15 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
uint64_t NewGuestSP = OldGuestSP;
|
||||
|
||||
// Pulling from context here
|
||||
bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
const bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
const uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (SRAEnabled) {
|
||||
if (IsAddressInJITCode(OldPC, false)) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
@@ -324,11 +368,11 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
#endif
|
||||
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(ucontext, IgnoreMask);
|
||||
SpillSRA(Thread, ucontext, IgnoreMask);
|
||||
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
|
||||
} else {
|
||||
if (!IsAddressInJITCode(OldPC, true)) {
|
||||
if (!IsAddressInDispatcher(OldPC)) {
|
||||
// This is likely to cause issues but in some cases it isn't fatal
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
@@ -374,8 +418,13 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
if (GuestAction->sa_flags & SA_SIGINFO) {
|
||||
// Setup ucontext a bit
|
||||
if (Is64BitMode) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::_libc_fpstate));
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86_64::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
@@ -397,8 +446,9 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
@@ -410,11 +460,12 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
@@ -443,9 +494,21 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
@@ -470,8 +533,13 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
else {
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
@@ -494,21 +562,23 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
@@ -528,15 +598,26 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
@@ -619,14 +700,14 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
else {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SignalReturn;
|
||||
LOGMAN_THROW_A_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
// This state will be restored on rt_sigreturn
|
||||
memset(Frame->State.xmm, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.xmm.avx.data, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
|
||||
Frame->State.FCW = 0x37F;
|
||||
Frame->State.FTW = 0xFFFF;
|
||||
@@ -634,44 +715,44 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
|
||||
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Signal, ucontext);
|
||||
StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
@@ -682,9 +763,9 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -695,16 +776,16 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
|
||||
|
||||
// Our ref counting doesn't matter anymore
|
||||
SignalHandlerRefCounter = 0;
|
||||
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
|
||||
|
||||
// Set the new PC
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
@@ -713,24 +794,24 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (ThreadState->RunningEvents.ThreadSleeping) {
|
||||
if (Thread->RunningEvents.ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
++ThreadState->CTX->IdleWaitRefCount;
|
||||
++Thread->CTX->IdleWaitRefCount;
|
||||
}
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -749,31 +830,4 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
for (auto iter = CodeBuffers.begin(); iter != CodeBuffers.end(); ++iter) {
|
||||
auto [start, end] = *iter;
|
||||
if (start == reinterpret_cast<uint64_t>(start_to_remove)) {
|
||||
CodeBuffers.erase(iter);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher, bool IncludeCompileService) const {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (IncludeDispatcher && IsAddressInDispatcher(Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (IncludeCompileService && ThreadState->CompileService && ThreadState->CompileService->IsAddressInJITCode(Address)) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
+47
-42
@@ -1,8 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <cstdint>
|
||||
@@ -21,22 +19,20 @@ struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
struct DispatcherConfig {
|
||||
bool ExecuteBlocksWithCall = false;
|
||||
uintptr_t ExitFunctionLink = 0;
|
||||
uintptr_t ExitFunctionLinkThis = 0;
|
||||
bool StaticRegisterAssignment = false;
|
||||
bool StaticRegisterAllocation = false;
|
||||
};
|
||||
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
CPUBackend::AsmDispatch DispatchPtr;
|
||||
CPUBackend::JITCallback CallbackPtr;
|
||||
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
|
||||
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -48,61 +44,70 @@ public:
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
uint64_t GuestSignal_SIGILL{};
|
||||
uint64_t GuestSignal_SIGTRAP{};
|
||||
uint64_t GuestSignal_SIGSEGV{};
|
||||
uint64_t IntCallbackReturnAddress{};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
struct SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
} SynchronousFaultData;
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(int Signal, void *info, void *ucontext);
|
||||
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
|
||||
void RegisterCodeBuffer(uint8_t* start, size_t size) {
|
||||
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
|
||||
reinterpret_cast<uint64_t>(start + size));
|
||||
}
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const;
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ThreadState {Thread} {}
|
||||
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(int Signal, void *ucontext);
|
||||
void RestoreThreadState(void *ucontext);
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
static constexpr size_t MaxInterpreterTrampolineSize = 128;
|
||||
|
||||
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
|
||||
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
, config {Config}
|
||||
{}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
|
||||
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
|
||||
bool SRAEnabled = false;
|
||||
virtual void SpillSRA(void *ucontext, uint32_t IgnoreMask) {}
|
||||
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
DispatcherConfig config;
|
||||
|
||||
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
|
||||
static uint64_t GetCompileBlockPtr();
|
||||
|
||||
private:
|
||||
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
|
||||
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
};
|
||||
|
||||
}
|
||||
+209
-85
@@ -18,21 +18,26 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include "xbyak/xbyak.h"
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE r14
|
||||
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread)
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: Dispatcher(ctx, config)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
|
||||
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
|
||||
nullptr) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
|
||||
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
|
||||
DispatchPtr = getCurr<AsmDispatch>();
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -78,11 +83,10 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
|
||||
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
|
||||
|
||||
Label LoopTop;
|
||||
Label FullLookup;
|
||||
Label CallBlock;
|
||||
Label NoBlock;
|
||||
Label ExitBlock;
|
||||
Label ThreadPauseHandler;
|
||||
@@ -92,29 +96,24 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
{
|
||||
// Load our RIP
|
||||
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
|
||||
mov(rdx, qword STATE_PTR(CPUState, rip));
|
||||
|
||||
// L1 Cache
|
||||
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rax, rdx);
|
||||
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
cmp(qword[r13 + rax + 8], rdx);
|
||||
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
|
||||
jne(FullLookup);
|
||||
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(qword[r13 + rax + 0]);
|
||||
} else {
|
||||
mov(rax, qword[r13 + rax + 0]);
|
||||
jmp(CallBlock);
|
||||
}
|
||||
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
|
||||
|
||||
L(FullLookup);
|
||||
mov(r13, Thread->LookupCache->GetPagePointer());
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Full lookup
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
mov(rax, rdx);
|
||||
mov(rbx, VirtualMemorySize - 1);
|
||||
and_(rax, rbx);
|
||||
@@ -143,7 +142,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
je(NoBlock);
|
||||
|
||||
// Update L1
|
||||
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rcx, 1);
|
||||
@@ -151,30 +150,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
mov(qword[r13 + rcx*8 + 0], rax);
|
||||
|
||||
// Real block if we made it here
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(rax);
|
||||
} else {
|
||||
L(CallBlock);
|
||||
mov(rdi, STATE);
|
||||
call(rax);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(LoopTop);
|
||||
// Else we need to pause now
|
||||
jmp(ThreadPauseHandler);
|
||||
ud2();
|
||||
}
|
||||
else {
|
||||
jmp(LoopTop);
|
||||
}
|
||||
}
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -193,10 +169,37 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
mov(rsi, STATE);
|
||||
@@ -204,20 +207,84 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
call(rax);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, config.ExitFunctionLinkThis);
|
||||
mov(rsi, STATE);
|
||||
mov(rdx, rax); // rax is set at the block end
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
mov(rax, config.ExitFunctionLink);
|
||||
call(rax);
|
||||
jmp(rax);
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
mov(rsi, rax); // rax is set at the block end
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else {
|
||||
jmp(rax);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
@@ -237,7 +304,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
}
|
||||
|
||||
{
|
||||
CallbackPtr = getCurr<CPUBackend::JITCallback>();
|
||||
CallbackPtr = getCurr<JITCallback>();
|
||||
|
||||
push(rbx);
|
||||
push(rbp);
|
||||
@@ -252,7 +319,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
// XXX: XMM?
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
|
||||
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
@@ -260,12 +327,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
|
||||
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
|
||||
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
mov(qword [rbx], rax);
|
||||
|
||||
// Store RIP to the context state
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
|
||||
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
|
||||
|
||||
// Back to the loop top now
|
||||
jmp(LoopTop);
|
||||
@@ -280,30 +347,34 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = getCurr<uint64_t>();
|
||||
GuestSignal_SIGILL = getCurr<uint64_t>();
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = getCurr<uint64_t>();
|
||||
GuestSignal_SIGTRAP = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
int3();
|
||||
}
|
||||
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
add(byte [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)], 1);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)], X86State::X86_TRAPNO_OF);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)], 0);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)], 0x80);
|
||||
{
|
||||
// Guest SIGSEGV handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
hlt();
|
||||
}
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
IntCallbackReturnAddress = getCurr<uint64_t>();
|
||||
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
|
||||
// rdi = thread
|
||||
@@ -337,39 +408,92 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
|
||||
}
|
||||
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.X86;
|
||||
}
|
||||
|
||||
Pointers.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Pointers.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Pointers.ThreadStopHandler = ThreadStopHandlerAddress;
|
||||
Pointers.ThreadPauseHandler = ThreadPauseHandlerAddress;
|
||||
Pointers.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
|
||||
Pointers.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
|
||||
Pointers.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Pointers.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
|
||||
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
|
||||
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
emit.je(RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.mov(rax, GuestRIP);
|
||||
emit.mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
|
||||
|
||||
// Stop the thread
|
||||
emit.mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.jmp(rax);
|
||||
}
|
||||
|
||||
emit.L(RunBlock);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
|
||||
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
Label InlineIRData;
|
||||
|
||||
emit.mov(rdi, STATE);
|
||||
emit.lea(rsi, ptr[rip + InlineIRData]);
|
||||
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
|
||||
emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
|
||||
emit.L(InlineIRData);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<X86Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -17,7 +17,10 @@ namespace FEXCore::CPU {
|
||||
|
||||
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
virtual ~X86Dispatcher() override;
|
||||
};
|
||||
|
||||
+52
-30
@@ -19,6 +19,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
@@ -177,22 +178,17 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
|
||||
Decoder::Decoder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN } {
|
||||
// Using mmap is a start-up time optimization
|
||||
// Take advantage of page faulting to reduce startup time for minimal runtime cost
|
||||
DecodedBuffer =
|
||||
reinterpret_cast<FEXCore::X86Tables::DecodedInst *>(
|
||||
FEXCore::Allocator::mmap(0, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize,
|
||||
PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN }
|
||||
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
}
|
||||
|
||||
Decoder::~Decoder() {
|
||||
FEXCore::Allocator::munmap(DecodedBuffer, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize);
|
||||
PoolObject.UnclaimBuffer();
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
@@ -204,14 +200,7 @@ uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
}
|
||||
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
if (Size == 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (Size > sizeof(uint64_t)) {
|
||||
LOGMAN_MSG_A_FMT("Unknown data size to read");
|
||||
return 0;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
@@ -351,13 +340,15 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
if (Displacement) {
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
else if (ModRM.mod == 0) {
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
@@ -408,7 +399,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
|
||||
uint8_t DestSize{};
|
||||
@@ -532,7 +523,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
@@ -641,7 +632,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
@@ -666,7 +657,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
@@ -692,7 +683,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
"REX PREFIX should have been decoded before this!");
|
||||
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
@@ -749,7 +740,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
|
||||
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -1140,7 +1131,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1148,6 +1139,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// XXX: Load symbol data
|
||||
SymbolAvailable = false;
|
||||
@@ -1169,6 +1161,13 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
// Entry is a jump target
|
||||
BlocksToDecode.emplace(PC);
|
||||
|
||||
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
|
||||
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1185,10 +1184,33 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpMinAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
|
||||
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpMaxPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMaxPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", PC + PCOffset, PC);
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
|
||||
+7
-1
@@ -27,7 +27,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
@@ -38,6 +38,11 @@ public:
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
private:
|
||||
// To pass any information from instruction prefixes
|
||||
// down into the actual instruction handling machinery.
|
||||
@@ -63,6 +68,7 @@ private:
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
uint8_t const *InstStream;
|
||||
|
||||
+588
-432
File diff suppressed because it is too large.
Load diff
+14
-1
@@ -6,8 +6,10 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
@@ -27,6 +29,10 @@ public:
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
|
||||
void AlertLibrariesChanged() {
|
||||
LibraryMapChanged = true;
|
||||
}
|
||||
|
||||
private:
|
||||
void Break(int signal);
|
||||
|
||||
@@ -38,6 +44,9 @@ private:
|
||||
|
||||
void SendACK(std::ostream &stream, bool NACK);
|
||||
|
||||
Event ThreadBreakEvent{};
|
||||
void WaitForThreadWakeup();
|
||||
|
||||
struct HandledPacketType {
|
||||
std::string Response{};
|
||||
enum ResponseType {
|
||||
@@ -74,9 +83,13 @@ private:
|
||||
bool NoAckMode{false};
|
||||
bool NonStopMode{false};
|
||||
std::string ThreadString{};
|
||||
std::string MemoryMapString{};
|
||||
std::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
|
||||
+28
-5
@@ -1,5 +1,5 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
@@ -59,11 +59,16 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
@@ -83,6 +88,24 @@ HostFeatures::HostFeatures() {
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
|
||||
+213
-211
@@ -20,7 +20,7 @@ DEF_OP(TruncElementPair) {
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: {
|
||||
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Pair);
|
||||
uint64_t Result{};
|
||||
Result = Src[0] & ~0U;
|
||||
Result |= Src[1] << 32;
|
||||
@@ -69,11 +69,11 @@ DEF_OP(CycleCounter) {
|
||||
|
||||
DEF_OP(Add) {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a + b; };
|
||||
auto *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
auto *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a + b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(4, uint32_t, Func)
|
||||
@@ -84,11 +84,11 @@ DEF_OP(Add) {
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a - b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a - b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(4, uint32_t, Func)
|
||||
@@ -99,9 +99,9 @@ DEF_OP(Sub) {
|
||||
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = -static_cast<int32_t>(Src);
|
||||
@@ -115,10 +115,10 @@ DEF_OP(Neg) {
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -138,10 +138,10 @@ DEF_OP(Mul) {
|
||||
|
||||
DEF_OP(UMul) {
|
||||
auto Op = IROp->C<IR::IROp_UMul>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -161,9 +161,9 @@ DEF_OP(UMul) {
|
||||
|
||||
DEF_OP(Div) {
|
||||
auto Op = IROp->C<IR::IROp_Div>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -179,7 +179,7 @@ DEF_OP(Div) {
|
||||
GD = static_cast<int64_t>(Src1) / static_cast<int64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Src1) / *GetSrc<__int128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -189,10 +189,10 @@ DEF_OP(Div) {
|
||||
|
||||
DEF_OP(UDiv) {
|
||||
auto Op = IROp->C<IR::IROp_UDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -208,7 +208,7 @@ DEF_OP(UDiv) {
|
||||
GD = static_cast<uint64_t>(Src1) / static_cast<uint64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src1) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -218,10 +218,10 @@ DEF_OP(UDiv) {
|
||||
|
||||
DEF_OP(Rem) {
|
||||
auto Op = IROp->C<IR::IROp_Rem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -237,7 +237,7 @@ DEF_OP(Rem) {
|
||||
GD = static_cast<int64_t>(Src1) % static_cast<int64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Src1) % *GetSrc<__int128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -247,10 +247,10 @@ DEF_OP(Rem) {
|
||||
|
||||
DEF_OP(URem) {
|
||||
auto Op = IROp->C<IR::IROp_URem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -266,7 +266,7 @@ DEF_OP(URem) {
|
||||
GD = static_cast<uint64_t>(Src1) % static_cast<uint64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src1) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -276,10 +276,10 @@ DEF_OP(URem) {
|
||||
|
||||
DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
@@ -298,10 +298,10 @@ DEF_OP(MulH) {
|
||||
|
||||
DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint64_t>(Src1) * static_cast<uint64_t>(Src2);
|
||||
@@ -324,11 +324,11 @@ DEF_OP(UMulH) {
|
||||
|
||||
DEF_OP(Or) {
|
||||
auto Op = IROp->C<IR::IROp_Or>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a | b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a | b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -342,11 +342,11 @@ DEF_OP(Or) {
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a & b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a & b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -361,8 +361,8 @@ DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
constexpr auto Func = [](auto a, auto b) {
|
||||
using Type = decltype(a);
|
||||
return static_cast<Type>(a & static_cast<Type>(~b));
|
||||
@@ -379,11 +379,11 @@ DEF_OP(Andn) {
|
||||
|
||||
DEF_OP(Xor) {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a ^ b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a ^ b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -396,11 +396,11 @@ DEF_OP(Xor) {
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint32_t>(Src1) << (Src2 & Mask);
|
||||
@@ -414,11 +414,11 @@ DEF_OP(Lshl) {
|
||||
|
||||
DEF_OP(Lshr) {
|
||||
auto Op = IROp->C<IR::IROp_Lshr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint32_t>(Src1) >> (Src2 & Mask);
|
||||
@@ -432,11 +432,11 @@ DEF_OP(Lshr) {
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
auto Op = IROp->C<IR::IROp_Ashr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = (uint32_t)(static_cast<int32_t>(Src1) >> (Src2 & Mask));
|
||||
@@ -450,12 +450,12 @@ DEF_OP(Ashr) {
|
||||
|
||||
DEF_OP(Ror) {
|
||||
auto Op = IROp->C<IR::IROp_Ror>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Ror = [] (auto In, auto R) {
|
||||
auto RotateMask = sizeof(In) * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const auto Ror = [] (auto In, auto R) {
|
||||
const auto RotateMask = sizeof(In) * 8 - 1;
|
||||
R &= RotateMask;
|
||||
return (In >> R) | (In << (sizeof(In) * 8 - R));
|
||||
};
|
||||
@@ -474,11 +474,11 @@ DEF_OP(Ror) {
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
|
||||
__uint128_t Result{};
|
||||
Result = Src1;
|
||||
Result <<= sizeof(Src1) * 8;
|
||||
@@ -500,7 +500,7 @@ DEF_OP(Extr) {
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto Op = IROp->C<IR::IROp_PDep>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (OpSize != 4 && OpSize != 8) {
|
||||
@@ -508,10 +508,10 @@ DEF_OP(PDep) {
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Input)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Input);
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Mask)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Mask);
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Index = 0; Mask > 0; Index++) {
|
||||
@@ -532,10 +532,10 @@ DEF_OP(PExt) {
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Input)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Input);
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Mask)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Mask);
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Offset = 0; Mask > 0; Offset++) {
|
||||
@@ -549,39 +549,39 @@ DEF_OP(PExt) {
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
int32_t Res = Source / Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const int32_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
int64_t Res = Source / Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const int64_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __int128_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
@@ -593,39 +593,39 @@ DEF_OP(LDiv) {
|
||||
|
||||
DEF_OP(LUDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
uint32_t Res = Source / Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const uint32_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
uint64_t Res = Source / Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const uint64_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __uint128_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
@@ -637,39 +637,39 @@ DEF_OP(LUDiv) {
|
||||
|
||||
DEF_OP(LRem) {
|
||||
auto Op = IROp->C<IR::IROp_LRem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit Remainder from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
int32_t Res = Source % Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const int32_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
int64_t Res = Source % Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const int64_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __int128_t Res = Source % Divisor;
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
break;
|
||||
@@ -680,39 +680,39 @@ DEF_OP(LRem) {
|
||||
|
||||
DEF_OP(LURem) {
|
||||
auto Op = IROp->C<IR::IROp_LURem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit Remainder from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
uint32_t Res = Source % Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const uint32_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
uint64_t Res = Source % Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const uint64_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source % Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __uint128_t Res = Source % Divisor;
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
break;
|
||||
@@ -723,62 +723,62 @@ DEF_OP(LURem) {
|
||||
|
||||
DEF_OP(Not) {
|
||||
auto Op = IROp->C<IR::IROp_Not>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t mask[9]= { 0, 0xFF, 0xFFFF, 0, 0xFFFFFFFF, 0, 0, 0, 0xFFFFFFFFFFFFFFFFULL };
|
||||
uint64_t Mask = mask[OpSize];
|
||||
const uint64_t Mask = mask[OpSize];
|
||||
GD = (~Src) & Mask;
|
||||
}
|
||||
|
||||
DEF_OP(Popcount) {
|
||||
auto Op = IROp->C<IR::IROp_Popcount>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::popcount(Src);
|
||||
}
|
||||
|
||||
DEF_OP(FindLSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindLSB>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Result = FindFirstSetBit(Src);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t Result = FindFirstSetBit(Src);
|
||||
GD = Result - 1;
|
||||
}
|
||||
|
||||
DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FindMSB size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FindTrailingZeros) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeros>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
@@ -788,26 +788,26 @@ DEF_OP(FindTrailingZeros) {
|
||||
|
||||
DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
@@ -817,12 +817,12 @@ DEF_OP(CountLeadingZeroes) {
|
||||
|
||||
DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Src)); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Src)); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Src)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown REV size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
@@ -830,34 +830,36 @@ DEF_OP(Rev) {
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
if (Op->Width == 64) {
|
||||
SourceMask = ~0ULL;
|
||||
uint64_t DestMask = ~(SourceMask << Op->lsb);
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
|
||||
}
|
||||
const uint64_t DestMask = ~(SourceMask << Op->lsb);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Dest);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
if (Op->Width == 64) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
SourceMask <<= Op->lsb;
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = (Src & SourceMask) >> Op->lsb;
|
||||
}
|
||||
|
||||
DEF_OP(Sbfe) {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for SBFE: {}", IROp->Size);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for SBFE: {}", IROp->Size);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
const uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
Src <<= ShiftLeftAmount;
|
||||
Src >>= ShiftRightAmount;
|
||||
GD = Src;
|
||||
@@ -865,20 +867,20 @@ DEF_OP(Sbfe) {
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
uint64_t ArgTrue;
|
||||
uint64_t ArgFalse;
|
||||
|
||||
if (OpSize == 4) {
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[3]);
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->FalseVal);
|
||||
} else {
|
||||
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[3]);
|
||||
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->FalseVal);
|
||||
}
|
||||
|
||||
bool CompResult;
|
||||
@@ -894,9 +896,9 @@ DEF_OP(Select) {
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
|
||||
uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Header.Args[0]);
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
@@ -904,7 +906,7 @@ DEF_OP(VExtractToGPR) {
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
@@ -915,7 +917,7 @@ DEF_OP(VExtractToGPR) {
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
@@ -924,25 +926,25 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
DEF_OP(Float_ToGPR_ZS) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
@@ -951,25 +953,25 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
|
||||
DEF_OP(Float_ToGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
@@ -980,9 +982,9 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
uint32_t ResultFlags{};
|
||||
if (Op->ElementSize == 4) {
|
||||
float Src1 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
float Src2 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
const float Src1 = *GetSrc<float*>(Data->SSAData, Op->Scalar1);
|
||||
const float Src2 = *GetSrc<float*>(Data->SSAData, Op->Scalar2);
|
||||
const bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
if (Unordered || (Src1 < Src2)) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
@@ -1000,9 +1002,9 @@ DEF_OP(FCmp) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Scalar1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Scalar2);
|
||||
const bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
if (Unordered || (Src1 < Src2)) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -25,20 +26,13 @@ static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
SignalReturn(Data->State);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
Data->State->CTX->InterpreterCallbackReturn(Data->State, Data->StackEntry);
|
||||
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
|
||||
}
|
||||
|
||||
DEF_OP(ExitFunction) {
|
||||
@@ -48,7 +42,7 @@ DEF_OP(ExitFunction) {
|
||||
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
|
||||
|
||||
void *ContextData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->NewRIP);
|
||||
|
||||
memcpy(ContextData, Src, OpSize);
|
||||
|
||||
@@ -57,22 +51,22 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->Header.Args[0]);
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TargetBlock);
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
bool CompResult;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
@@ -133,7 +127,7 @@ DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->Header.Args[0]));
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
@@ -147,15 +141,15 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
Data->State->CTX->RemoveCodeEntry(Data->State, Data->CurrentEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
|
||||
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
|
||||
|
||||
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
|
||||
@@ -14,10 +14,10 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
@@ -35,31 +35,31 @@ DEF_OP(VInsGPR) {
|
||||
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), Op->Header.ElementSize);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
@@ -68,15 +68,15 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 8);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 4);
|
||||
break;
|
||||
}
|
||||
@@ -86,14 +86,14 @@ DEF_OP(Float_FToF) {
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
@@ -104,14 +104,14 @@ DEF_OP(Vector_SToF) {
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -122,14 +122,14 @@ DEF_OP(Vector_FToZS) {
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -140,14 +140,14 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- float
|
||||
// Only the lower elements from the source
|
||||
@@ -172,17 +172,17 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
const auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
const auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
|
||||
@@ -360,7 +360,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
|
||||
// Pseudo-code
|
||||
// Dst = InvMixColumns(STATE)
|
||||
@@ -371,8 +371,8 @@ DEF_OP(AESImc) {
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -391,8 +391,8 @@ DEF_OP(AESEnc) {
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -409,8 +409,8 @@ DEF_OP(AESEncLast) {
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -429,8 +429,8 @@ DEF_OP(AESDec) {
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -447,7 +447,7 @@ DEF_OP(AESDecLast) {
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
// Pseudo-code
|
||||
// X3 = Src1[127:96]
|
||||
@@ -513,6 +513,44 @@ DEF_OP(CRC32) {
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
const auto Selector = Op->Selector;
|
||||
auto* Dst = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
auto* Src1 = GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
auto* Src2 = GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
const uint64_t TMP1 = (Selector & 0x01) == 0 ? Src1[0] : Src1[1];
|
||||
const uint64_t TMP2 = (Selector & 0x10) == 0 ? Src2[0] : Src2[1];
|
||||
|
||||
const auto make_lo = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 0; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs << i;
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
const auto make_hi = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 1; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs >> (64 - i);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
|
||||
Dst[0] = make_lo(TMP1, TMP2);
|
||||
Dst[1] = make_hi(TMP1, TMP2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+134
-72
@@ -20,99 +20,90 @@ DEF_OP(F80LOADFCW) {
|
||||
|
||||
DEF_OP(F80ADD) {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SUB) {
|
||||
auto Op = IROp->C<IR::IROp_F80Sub>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80MUL) {
|
||||
auto Op = IROp->C<IR::IROp_F80Mul>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80DIV) {
|
||||
auto Op = IROp->C<IR::IROp_F80Div>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F80FYL2X>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80ATAN>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM1>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F80SCALE>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CVT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVT>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
@@ -131,9 +122,9 @@ DEF_OP(F80CVT) {
|
||||
|
||||
DEF_OP(F80CVTINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
@@ -160,13 +151,13 @@ DEF_OP(F80CVTTO) {
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->Header.Args[0]);
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
@@ -180,13 +171,13 @@ DEF_OP(F80CVTTOINT) {
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
@@ -197,77 +188,73 @@ DEF_OP(F80CVTTOINT) {
|
||||
|
||||
DEF_OP(F80ROUND) {
|
||||
auto Op = IROp->C<IR::IROp_F80Round>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80F2XM1>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::F2XM1(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::F2XM1(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80TAN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FTAN(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FTAN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SQRT) {
|
||||
auto Op = IROp->C<IR::IROp_F80SQRT>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSQRT(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSQRT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F80SIN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSIN(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSIN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80COS) {
|
||||
auto Op = IROp->C<IR::IROp_F80COS>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FCOS(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FCOS(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_EXP) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_SIG) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CMP) {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
uint32_t ResultFlags{};
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
bool eq, lt, nan;
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
@@ -288,7 +275,7 @@ DEF_OP(F80CMP) {
|
||||
|
||||
DEF_OP(F80BCDLOAD) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
@@ -323,7 +310,7 @@ DEF_OP(F80BCDLOAD) {
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]));
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
@@ -356,6 +343,81 @@ DEF_OP(F80BCDSTORE) {
|
||||
memcpy(GDP, BCD, 10);
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = sin(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = cos(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = tan(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64F2XM1>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = exp2(Src) - 1.0;
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64ATAN>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = atan2(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = fmod(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = remainder(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F64FYL2X>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = Src2 * log2(Src1);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double trunc = (double)(int64_t)(Src2); //truncate
|
||||
const double Tmp = Src1 * exp2(trunc);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -222,6 +222,73 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(double src) {
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(double src) {
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(double src) {
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(double src) {
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(double src1, double src2) {
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(double src1, double src2) {
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(double src1, double src2) {
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(double src1, double src2) {
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(double src1, double src2) {
|
||||
double trunc = (double)(int64_t)(src2); //truncate
|
||||
return src1 * exp2(trunc);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
|
||||
@@ -14,7 +14,7 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]) >> Op->Flag) & 1;
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Value) >> Op->Flag) & 1;
|
||||
}
|
||||
#undef DEF_OP
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Dispatcher;
|
||||
class X86DispatchGenerator;
|
||||
class Arm64DispatchGenerator;
|
||||
|
||||
@@ -20,28 +21,27 @@ using DestMapType = std::vector<uint32_t>;
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
explicit InterpreterCore(Dispatcher *Dispatch,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *State;
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher{};
|
||||
size_t BufferUsed;
|
||||
Dispatcher *Dispatch;
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
|
||||
@@ -9,66 +9,101 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <memory>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
|
||||
#include "InterpreterOps.h"
|
||||
|
||||
#if defined(_M_X86_64)
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#elif defined(_M_ARM_64)
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#else
|
||||
#error missing arch
|
||||
#endif
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Dispatch(Dispatcher)
|
||||
{
|
||||
|
||||
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
|
||||
InterpreterOps::InterpretIR(Thread, Thread->CurrentFrame->State.rip, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
|
||||
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: CTX {ctx}
|
||||
, State {Thread} {
|
||||
|
||||
if (!CompileThread &&
|
||||
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
CreateAsmDispatch(ctx, Thread);
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
|
||||
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
|
||||
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
|
||||
|
||||
auto DestBuffer = BufferStart;
|
||||
|
||||
if (GDBEnabled) {
|
||||
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
|
||||
DestBuffer += GDBSize;
|
||||
BufferUsed += GDBSize;
|
||||
}
|
||||
|
||||
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
|
||||
DestBuffer += TrampolineSize;
|
||||
BufferUsed += TrampolineSize;
|
||||
|
||||
|
||||
IR->Serialize(DestBuffer);
|
||||
DestBuffer += IRSize;
|
||||
BufferUsed += IRSize;
|
||||
|
||||
return BufferStart;
|
||||
}
|
||||
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
void InterpreterCore::ClearCache() {
|
||||
// Calling this one is needed to setup the initial CurrentCodeBuffer
|
||||
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
BufferUsed = 0;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<InterpreterCore>(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
InterpreterCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
}
|
||||
@@ -12,9 +12,11 @@ namespace FEXCore::Core {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
struct DispatcherConfig;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -159,21 +159,26 @@
|
||||
break; \
|
||||
}
|
||||
|
||||
struct InterpVector256 {
|
||||
__uint128_t Lower;
|
||||
__uint128_t Upper;
|
||||
};
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.ID().Value];
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.Value];
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
|
||||
template<typename Res>
|
||||
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Src.ID().Value];
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Src.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
@@ -47,6 +47,16 @@ FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackH
|
||||
return {FABI_F64_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I16_F80, (void*)fn, HandlerIndex};
|
||||
@@ -122,6 +132,18 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
|
||||
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
|
||||
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
|
||||
|
||||
// Double Precision
|
||||
Info[Core::OPINDEX_F64SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle, Core::OPINDEX_F64SIN).fn);
|
||||
Info[Core::OPINDEX_F64COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle, Core::OPINDEX_F64COS).fn);
|
||||
Info[Core::OPINDEX_F64TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle, Core::OPINDEX_F64TAN).fn);
|
||||
Info[Core::OPINDEX_F64ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle, Core::OPINDEX_F64ATAN).fn);
|
||||
Info[Core::OPINDEX_F64F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle, Core::OPINDEX_F64F2XM1).fn);
|
||||
Info[Core::OPINDEX_F64FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle, Core::OPINDEX_F64FYL2X).fn);
|
||||
Info[Core::OPINDEX_F64FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle, Core::OPINDEX_F64FPREM).fn);
|
||||
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
|
||||
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
@@ -238,6 +260,12 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
@@ -261,6 +289,19 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
// Double Precision Unary
|
||||
COMMON_F64_OP(F2XM1)
|
||||
COMMON_F64_OP(TAN)
|
||||
COMMON_F64_OP(SIN)
|
||||
COMMON_F64_OP(COS)
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_F64_OP(FYL2X)
|
||||
COMMON_F64_OP(ATAN)
|
||||
COMMON_F64_OP(FPREM1)
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "InterpreterDefines.h"
|
||||
#include "InterpreterOps.h"
|
||||
#include "F80Ops.h"
|
||||
|
||||
@@ -112,8 +113,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -123,7 +122,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
|
||||
// Conversion ops
|
||||
@@ -166,6 +165,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, NoOp);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -176,6 +176,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
@@ -283,6 +284,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// F80 ops
|
||||
REGISTER_OP(F80LOADFCW, F80LOADFCW);
|
||||
@@ -311,6 +313,17 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(F80BCDLOAD, F80BCDLOAD);
|
||||
REGISTER_OP(F80BCDSTORE, F80BCDSTORE);
|
||||
|
||||
// F64 ops
|
||||
REGISTER_OP(F64SIN, F64SIN);
|
||||
REGISTER_OP(F64COS, F64COS);
|
||||
REGISTER_OP(F64TAN, F64TAN);
|
||||
REGISTER_OP(F64F2XM1, F64F2XM1);
|
||||
REGISTER_OP(F64ATAN, F64ATAN);
|
||||
REGISTER_OP(F64FPREM, F64FPREM);
|
||||
REGISTER_OP(F64FPREM1, F64FPREM1);
|
||||
REGISTER_OP(F64FYL2X, F64FYL2X);
|
||||
REGISTER_OP(F64SCALE, F64SCALE);
|
||||
|
||||
return Handlers;
|
||||
}();
|
||||
|
||||
@@ -321,38 +334,37 @@ void InterpreterOps::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, IROpData *Data
|
||||
void InterpreterOps::Op_NoOp(FEXCore::IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData) {
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *CurrentIR) {
|
||||
volatile void *StackEntry = alloca(0);
|
||||
|
||||
// Debug data is only passed in debug builds
|
||||
#ifndef NDEBUG
|
||||
// TODO: should be moved to an IR Op
|
||||
Thread->Stats.InstructionsExecuted.fetch_add(DebugData->GuestInstructionCount);
|
||||
#endif
|
||||
|
||||
uintptr_t ListSize = CurrentIR->GetSSACount();
|
||||
const uintptr_t ListSize = CurrentIR->GetSSACount();
|
||||
|
||||
static_assert(sizeof(FEXCore::IR::IROp_Header) == 4);
|
||||
static_assert(sizeof(FEXCore::IR::OrderedNode) == 16);
|
||||
|
||||
auto BlockEnd = CurrentIR->GetBlocks().end();
|
||||
|
||||
InterpreterOps::IROpData OpData{};
|
||||
OpData.State = Thread;
|
||||
OpData.SSAData = alloca(ListSize * 16);
|
||||
OpData.CurrentEntry = Entry;
|
||||
OpData.CurrentIR = CurrentIR;
|
||||
OpData.StackEntry = StackEntry;
|
||||
OpData.BlockIterator = CurrentIR->GetBlocks().begin();
|
||||
constexpr size_t ListEntrySizeInBytes = sizeof(InterpVector256);
|
||||
const size_t SSADataSize = ListSize * ListEntrySizeInBytes;
|
||||
|
||||
// Clear them all to zero. Required for Zero-extend semantics
|
||||
memset(OpData.SSAData, 0, ListSize * 16);
|
||||
InterpreterOps::IROpData OpData{
|
||||
.State = Frame->Thread,
|
||||
.CurrentEntry = Frame->State.rip,
|
||||
.CurrentIR = CurrentIR,
|
||||
.StackEntry = StackEntry,
|
||||
.SSAData = alloca(SSADataSize),
|
||||
.BlockResults = {},
|
||||
.BlockIterator = CurrentIR->GetBlocks().begin(),
|
||||
};
|
||||
|
||||
// Clear all SSAData entries to zero. Required for Zero-extend semantics
|
||||
memset(OpData.SSAData, 0, SSADataSize);
|
||||
|
||||
while (1) {
|
||||
using namespace FEXCore::IR;
|
||||
auto [BlockNode, BlockHeader] = OpData.BlockIterator();
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
// Reset the block results per block
|
||||
memset(&OpData.BlockResults, 0, sizeof(OpData.BlockResults));
|
||||
|
||||
@@ -28,6 +28,8 @@ namespace FEXCore::CPU {
|
||||
FABI_F80_I32,
|
||||
FABI_F32_F80,
|
||||
FABI_F64_F80,
|
||||
FABI_F64_F64,
|
||||
FABI_F64_F64_F64,
|
||||
FABI_I16_F80,
|
||||
FABI_I32_F80,
|
||||
FABI_I64_F80,
|
||||
@@ -45,14 +47,14 @@ namespace FEXCore::CPU {
|
||||
class InterpreterOps {
|
||||
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
uint64_t CurrentEntry{};
|
||||
FEXCore::IR::IRListView *CurrentIR{};
|
||||
FEXCore::IR::IRListView const *CurrentIR{};
|
||||
volatile void *StackEntry{};
|
||||
void *SSAData{};
|
||||
struct {
|
||||
@@ -140,8 +142,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -151,7 +151,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -197,6 +197,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -301,6 +302,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
|
||||
///< F80 ops
|
||||
DEF_OP(F80LOADFCW);
|
||||
@@ -328,6 +330,17 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(F80CMP);
|
||||
DEF_OP(F80BCDLOAD);
|
||||
DEF_OP(F80BCDSTORE);
|
||||
|
||||
//< F64 ops
|
||||
DEF_OP(F64SIN);
|
||||
DEF_OP(F64COS);
|
||||
DEF_OP(F64TAN);
|
||||
DEF_OP(F64F2XM1);
|
||||
DEF_OP(F64ATAN);
|
||||
DEF_OP(F64FPREM);
|
||||
DEF_OP(F64FPREM1);
|
||||
DEF_OP(F64FYL2X);
|
||||
DEF_OP(F64SCALE);
|
||||
#undef DEF_OP
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
@@ -394,7 +407,7 @@ namespace FEXCore::CPU {
|
||||
return CompResult;
|
||||
}
|
||||
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
|
||||
return IROp->Size;
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -58,7 +59,7 @@ DEF_OP(StoreContext) {
|
||||
ContextPtr += Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
@@ -72,7 +73,7 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
@@ -102,14 +103,14 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
}
|
||||
|
||||
@@ -133,7 +134,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
@@ -227,9 +228,9 @@ DEF_OP(StoreMem) {
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Header.Args[0]);
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[1]), 16);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
@@ -237,8 +238,8 @@ DEF_OP(VLoadMemElement) {
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Header.Args[0]); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Header.Args[1])[Op->Index], sizeof(y)); \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
|
||||
+31
-12
@@ -4,6 +4,8 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -44,14 +46,26 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Halt: // HLT
|
||||
StopThread(Data->State);
|
||||
|
||||
Data->State->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = 1;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.Signal = Op->Reason.Signal;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.TrapNo = Op->Reason.TrapNumber;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.err_code = Op->Reason.ErrorRegister;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.si_code = Op->Reason.si_code;
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
break;
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
case SIGTRAP:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGSEGV);
|
||||
break;
|
||||
default:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break Reason: {}", Op->Reason); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -86,7 +100,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
uint8_t GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->RoundMode);
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t HostRounding{};
|
||||
__asm volatile(R"(
|
||||
@@ -126,16 +140,16 @@ DEF_OP(SetRoundingMode) {
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (OpSize <= 8) {
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
|
||||
}
|
||||
else if (OpSize == 16) {
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src0 = Src;
|
||||
uint64_t Src1 = Src >> 64;
|
||||
const auto Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Value);
|
||||
const uint64_t Src0 = Src;
|
||||
const uint64_t Src1 = Src >> 64;
|
||||
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
|
||||
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
|
||||
}
|
||||
@@ -157,6 +171,11 @@ DEF_OP(RDRAND) {
|
||||
// Second result is if we managed to read a valid random number or not
|
||||
DstPtr[1] = Result == 8 ? 1 : 0;
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
// Nop implementation
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -14,15 +14,15 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
uintptr_t Src = GetSrc<uintptr_t>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = GetSrc<uintptr_t>(Data->SSAData, Op->Pair);
|
||||
memcpy(GDP,
|
||||
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Lower);
|
||||
const void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Upper);
|
||||
|
||||
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
|
||||
|
||||
@@ -32,9 +32,9 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), OpSize);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+396
-375
File diff suppressed because it is too large.
Load diff
+352
-247
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,131 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|arm64
|
||||
desc: relocation logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -509,10 +510,10 @@ DEF_OP(AtomicSwap) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
case 1: swpalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swpalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
+21
-29
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
@@ -19,13 +20,6 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -33,7 +27,7 @@ DEF_OP(SignalReturn) {
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalReturnHandler)));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)));
|
||||
br(x0);
|
||||
}
|
||||
|
||||
@@ -46,10 +40,10 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalHandlerRefCountPointer)));
|
||||
ldr(w2, MemOperand(x0));
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(x0));
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
@@ -73,7 +67,7 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress};
|
||||
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
|
||||
ldr(x0, &l_BranchHost);
|
||||
@@ -82,10 +76,10 @@ DEF_OP(ExitFunction) {
|
||||
place(&l_BranchHost);
|
||||
place(&l_BranchGuest);
|
||||
} else {
|
||||
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RipReg = GetReg<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -96,7 +90,7 @@ DEF_OP(ExitFunction) {
|
||||
br(x1);
|
||||
|
||||
bind(&FullLookup);
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.DispatcherLoopTop)));
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)));
|
||||
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
br(TMP1);
|
||||
}
|
||||
@@ -104,9 +98,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
@@ -199,8 +193,8 @@ DEF_OP(Syscall) {
|
||||
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
|
||||
}
|
||||
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SyscallHandlerFunc)));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
blr(x3);
|
||||
@@ -383,7 +377,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x0, GetReg<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
@@ -442,7 +436,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
@@ -452,7 +446,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Entry);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.RemoveCodeEntryFromJIT)));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
SpillStaticRegs();
|
||||
blr(x2);
|
||||
FillStaticRegs();
|
||||
@@ -469,10 +463,10 @@ DEF_OP(CPUID) {
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
@@ -489,8 +483,6 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -500,7 +492,7 @@ void Arm64JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -13,22 +13,22 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
@@ -39,18 +39,18 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()).W());
|
||||
break;
|
||||
case 8:
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()).X());
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()).X());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -58,22 +58,22 @@ DEF_OP(VCastFromGPR) {
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -81,14 +81,14 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Header.Args[0].ID()).S());
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Scalar.ID()).S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Header.Args[0].ID()).D());
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Scalar.ID()).D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
@@ -99,10 +99,10 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -112,10 +112,10 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -125,11 +125,11 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
@@ -142,11 +142,11 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2S());
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
@@ -159,50 +159,50 @@ DEF_OP(Vector_FToI) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -14,41 +14,41 @@ using namespace vixl::aarch64;
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Vector.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesmc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesimc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
@@ -59,7 +59,7 @@ DEF_OP(AESKeyGenAssist) {
|
||||
|
||||
// Do a "regular" AESE step
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Src.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
@@ -102,16 +102,45 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node).Q();
|
||||
auto Src1 = GetSrc(Op->Src1.ID()).V2D();
|
||||
auto Src2 = GetSrc(Op->Src2.ID()).V2D();
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
pmull(Dst, Src1, Src2);
|
||||
break;
|
||||
case 0b00000001:
|
||||
mov(VTMP1.V1D(), Src1, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src2);
|
||||
break;
|
||||
case 0b00010000:
|
||||
mov(VTMP1.V1D(), Src2, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src1);
|
||||
break;
|
||||
case 0b00010001:
|
||||
pmull2(Dst, Src1, Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -13,7 +13,7 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+221
-258
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -35,6 +36,10 @@ $end_info$
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace {
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
@@ -71,11 +76,6 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
}
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
@@ -93,7 +93,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -108,7 +108,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -127,7 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -152,7 +152,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
else {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -173,7 +173,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -192,7 +192,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -203,6 +203,43 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
@@ -211,7 +248,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -229,7 +266,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -247,7 +284,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -268,7 +305,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -286,7 +323,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -309,7 +346,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -325,88 +362,72 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
FEXCore::IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
FEXCore::Allocator::mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
// Process specific
|
||||
Pointers.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
Pointers.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
Pointers.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
Pointers.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Pointers.RemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit);
|
||||
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Pointers.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Pointers.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Pointers.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
|
||||
|
||||
// Thread Specific
|
||||
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
|
||||
}
|
||||
|
||||
{
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
}
|
||||
|
||||
// Can't allocate a code buffer until after dispatcher is created
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
*GetBuffer() = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
SetAllowAssembler(true);
|
||||
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, HostSupportsSVE{ctx->HostFeatures.SupportsAVX}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -447,95 +468,76 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
if (!CompileThread) {
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalReturnInstruction = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
|
||||
if (!Core->Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Core->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
SetAllowAssembler(true);
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::Arm64JITCore::";
|
||||
auto Buffer = GetBuffer();
|
||||
Buffer->EmitString(JITString);
|
||||
Buffer->Align();
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
auto Buffer = GetBuffer();
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
Buffer->Reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
InitialCodeBuffer.Size *= 1.5;
|
||||
InitialCodeBuffer.Size = std::min(InitialCodeBuffer.Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(InitialCodeBuffer.Size);
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = Arm64JITCore::AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
*Buffer = vixl::CodeBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
|
||||
@@ -550,12 +552,12 @@ template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg].W();
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg].W();
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -565,12 +567,12 @@ template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -591,12 +593,12 @@ std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JI
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -605,12 +607,12 @@ aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -666,13 +668,18 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x0, Entry);
|
||||
@@ -681,9 +688,9 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
// AAPCS64
|
||||
@@ -706,31 +713,11 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
auto GuestEntry = GetCursorAddress<uint64_t>();
|
||||
GuestEntry = GetCursorAddress<uint8_t *>();
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Thread))); // Get thread
|
||||
ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
cbz(w0, &RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
LoadConstant(x0, Entry);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
|
||||
// Stop the thread
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadPauseHandlerSpillSRA)));
|
||||
br(x0);
|
||||
}
|
||||
bind(&RunBlock);
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
GetBuffer()->CursorForward(GDBSize);
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -752,9 +739,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
@@ -769,10 +757,6 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
bind(&IsTarget->second);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({GetCursorAddress<uintptr_t>(), 0, IR->GetID(BlockNode)});
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
|
||||
@@ -782,7 +766,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.back().HostCodeSize = GetCursorAddress<uintptr_t>() - DebugData->Subblocks.back().HostCodeStart;
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -795,68 +782,44 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = GetCursorAddress<uint64_t>();
|
||||
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(GuestEntry), CodeEnd - reinterpret_cast<uint64_t>(GuestEntry));
|
||||
auto CodeEnd = GetCursorAddress<uint8_t *>();
|
||||
CPU.EnsureIAndDCacheCoherency(GuestEntry, CodeEnd - GuestEntry);
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->HostCodeSize = CodeEnd - GuestEntry;
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
this->IR = nullptr;
|
||||
|
||||
return reinterpret_cast<void*>(GuestEntry);
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
Arm64JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
+84
-60
@@ -6,17 +6,23 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
@@ -37,14 +43,8 @@ using namespace vixl::aarch64;
|
||||
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -52,7 +52,7 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -60,21 +60,16 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
}
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
const bool HostSupportsSVE{};
|
||||
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
Label *PendingTargetLabel;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
|
||||
@@ -145,46 +140,77 @@ private:
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
// We don't want to mvoe above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
#if DEBUG
|
||||
vixl::aarch64::Disassembler Disasm;
|
||||
#endif
|
||||
|
||||
static uint64_t ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalReturnInstruction{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
@@ -275,8 +301,6 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -286,7 +310,7 @@ private:
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -326,7 +350,7 @@ private:
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -336,6 +360,7 @@ private:
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -442,9 +467,8 @@ private:
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+122
-58
@@ -4,6 +4,8 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
@@ -58,26 +60,26 @@ DEF_OP(LoadContext) {
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
strb(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 2:
|
||||
strh(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
strh(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 4:
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
str(Src.B(), MemOperand(STATE, Op->Offset));
|
||||
@@ -104,7 +106,7 @@ DEF_OP(LoadRegister) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0])) / 8;
|
||||
auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
@@ -113,30 +115,32 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_32>(Node), reg.W());
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_64>(Node), reg);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
|
||||
@@ -145,17 +149,17 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(host.B(), guest.B());
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
fmov(host.H(), guest.H());
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
fmov(host.S(), guest.S());
|
||||
@@ -165,7 +169,7 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.D(), guest.D());
|
||||
@@ -175,13 +179,13 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.Q(), guest.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -189,7 +193,7 @@ DEF_OP(StoreRegister) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = Op->Offset / 8 - 1;
|
||||
auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
@@ -198,29 +202,31 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 32);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Op->Value.ID()).GetCode() != reg.GetCode())
|
||||
mov(reg, GetReg<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
|
||||
@@ -233,36 +239,36 @@ DEF_OP(StoreRegister) {
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
ins(guest.V8H(), regOffs/2, host.V8H(), 0);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
ins(guest.V4S(), regOffs/4, host.V4S(), 0);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
ins(guest.V2D(), regOffs / 8, host.V2D(), 0);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (guest.GetCode() != host.GetCode())
|
||||
mov(guest.Q(), host.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -349,11 +355,11 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto value = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto value = GetReg<RA_64>(Op->Value.ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -392,7 +398,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Header.Args[0].ID());
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -441,25 +447,25 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * 16;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
strh(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
strh(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -467,15 +473,15 @@ DEF_OP(SpillRegister) {
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()).S(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()).S(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()).D(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()).D(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -539,7 +545,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
}
|
||||
|
||||
MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
@@ -570,7 +576,7 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -617,13 +623,43 @@ DEF_OP(LoadMem) {
|
||||
DEF_OP(LoadMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("LoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "LoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldapurb(Dst, MemSrc);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
@@ -744,13 +780,41 @@ DEF_OP(StoreMem) {
|
||||
|
||||
DEF_OP(StoreMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("StoreMemTSO: No offset allowed");
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "StoreMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlurh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlur(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlur(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
@@ -934,7 +998,7 @@ DEF_OP(VStoreMemElement) {
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
@@ -949,7 +1013,7 @@ DEF_OP(CacheLineClear) {
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
|
||||
+50
-41
@@ -4,13 +4,21 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <syscall.h>
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, GetCursorAddress<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
@@ -29,43 +37,38 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
hlt(4);
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
ResetStack();
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.OverflowExceptionHandler)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
add(sp, TMP1, 0);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadStopHandlerSpillSRA)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_Interrupt3: { // INT3
|
||||
ResetStack();
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadPauseHandlerSpillSRA)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
ResetStack();
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.UnimplementedInstructionHandler)));
|
||||
br(TMP1);
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)));
|
||||
br(TMP1);
|
||||
break;
|
||||
default:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -97,7 +100,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetReg<RA_64>(Op->RoundMode.ID());
|
||||
|
||||
// Setup the rounding flags correctly
|
||||
and_(TMP1, Src, 0b11);
|
||||
@@ -132,15 +135,15 @@ DEF_OP(Print) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.PrintValue)));
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)));
|
||||
}
|
||||
else {
|
||||
fmov(x0, GetSrc(Op->Header.Args[0].ID()).V1D());
|
||||
fmov(x0, GetSrc(Op->Value.ID()).V1D());
|
||||
// Bug in vixl that source vector needs to b V1D rather than V2D?
|
||||
fmov(x1, GetSrc(Op->Header.Args[0].ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.PrintVectorValue)));
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -219,6 +222,10 @@ DEF_OP(RDRAND) {
|
||||
cset(Dst.second, Condition::ne);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
hint(SystemHint::YIELD);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -227,6 +234,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -237,6 +245,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -15,13 +15,13 @@ DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
@@ -40,15 +40,15 @@ DEF_OP(CreateElementPair) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetReg<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetReg<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Upper.ID());
|
||||
RegTmp = w0;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetReg<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Upper.ID());
|
||||
RegTmp = x0;
|
||||
break;
|
||||
}
|
||||
@@ -70,7 +70,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+579
-457
File diff suppressed because it is too large.
Load diff
+7
-4
@@ -14,10 +14,13 @@ namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures();
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+236
-238
File diff suppressed because it is too large.
Load diff
+18
-26
@@ -29,13 +29,6 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
@@ -43,7 +36,7 @@ DEF_OP(SignalReturn) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
}
|
||||
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalReturnHandler)]);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
@@ -53,7 +46,7 @@ DEF_OP(CallbackReturn) {
|
||||
}
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)], 1);
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 8);
|
||||
@@ -91,14 +84,15 @@ DEF_OP(ExitFunction) {
|
||||
jmp(qword[rax]);
|
||||
|
||||
L(l_BranchHost);
|
||||
dq(ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress);
|
||||
//FEX_TODO(this is not per thread)
|
||||
dq(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
L(l_BranchGuest);
|
||||
dq(NewRIP);
|
||||
} else {
|
||||
Xbyak::Reg RipReg = GetSrc<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
mov(rcx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
|
||||
mov(rcx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)]);
|
||||
|
||||
mov(rax, RipReg);
|
||||
|
||||
@@ -113,7 +107,7 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
L(FullLookup);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], RipReg);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.DispatcherLoopTop)]);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)]);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
@@ -123,9 +117,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
@@ -181,13 +175,13 @@ DEF_OP(Syscall) {
|
||||
}
|
||||
|
||||
mov(rsi, STATE); // Move thread in to rsi
|
||||
mov(rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SyscallHandlerObj)]);
|
||||
mov(rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)]);
|
||||
mov(rdx, rsp);
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SyscallHandlerFunc)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -215,7 +209,7 @@ DEF_OP(Thunk) {
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
mov(rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rdi, GetSrc<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
|
||||
@@ -259,7 +253,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -272,7 +266,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
mov(rax, Entry); // imm64 move
|
||||
mov(rsi, rax);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.RemoveCodeEntryFromJIT)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -294,9 +288,9 @@ DEF_OP(CPUID) {
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (edx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.CPUIDObj)]);
|
||||
mov (edx, GetSrc<RA_32>(Op->Leaf.ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
@@ -304,7 +298,7 @@ DEF_OP(CPUID) {
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.CPUIDFunction)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -320,8 +314,6 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -330,7 +322,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -45,18 +45,18 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
movzx(rax, GetSrc<RA_8>(Op->Header.Args[0].ID()));
|
||||
movzx(rax, GetSrc<RA_8>(Op->Src.ID()));
|
||||
vmovq(GetDst(Node), rax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(rax, GetSrc<RA_16>(Op->Header.Args[0].ID()));
|
||||
movzx(rax, GetSrc<RA_16>(Op->Src.ID()));
|
||||
vmovq(GetDst(Node), rax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()).cvt32());
|
||||
vmovd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()).cvt32());
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()).cvt64());
|
||||
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()).cvt64());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown VCastFromGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -64,22 +64,23 @@ DEF_OP(VCastFromGPR) {
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -87,14 +88,15 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtss2sd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtss2sd(GetDst(Node), GetSrc(Op->Scalar.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtsd2ss(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtsd2ss(GetDst(Node), GetSrc(Op->Scalar.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Float_FToF sizes: 0x{:x}", Conv);
|
||||
@@ -105,7 +107,7 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
@@ -113,8 +115,8 @@ DEF_OP(Vector_SToF) {
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
@@ -127,10 +129,10 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -140,10 +142,10 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -151,15 +153,15 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
@@ -190,10 +192,10 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,44 +17,44 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
vaesimc(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
vaesimc(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
vaesenc(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesenc(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
vaesdec(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesdec(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->RCON);
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Src.ID()), Op->RCON);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (IROp->Size) {
|
||||
case 4:
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Src2.ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Src1.ID()));
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Src2.ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Src1.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", IROp->Size);
|
||||
}
|
||||
@@ -75,16 +75,37 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node);
|
||||
auto Src1 = GetSrc(Op->Src1.ID());
|
||||
auto Src2 = GetSrc(Op->Src2.ID());
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
case 0b00000001:
|
||||
case 0b00010000:
|
||||
case 0b00010001:
|
||||
vpclmulqdq(Dst, Src1, Src2, Op->Selector);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -18,7 +18,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
shr(rax, Op->Flag);
|
||||
and_(rax, 1);
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
|
||||
+139
-193
@@ -25,6 +25,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -43,6 +44,9 @@ $end_info$
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
|
||||
namespace {
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
@@ -55,31 +59,6 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
FEXCore::Allocator::mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
void X86JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
}
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
@@ -120,7 +99,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_VOID_U16: {
|
||||
PushRegs();
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
break;
|
||||
@@ -129,7 +108,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
movss(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -143,7 +122,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -158,7 +137,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -174,7 +153,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -188,7 +167,34 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
movsd(xmm1, GetSrc(IROp->Args[1].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -202,7 +208,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -215,7 +221,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -228,7 +234,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -244,7 +250,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -257,7 +263,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -275,7 +281,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -288,48 +294,42 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
|
||||
record[0] = HostCode;
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread)
|
||||
: CodeGenerator(Buffer.Size, Buffer.Ptr, nullptr)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread}
|
||||
, InitialCodeBuffer {Buffer}
|
||||
{
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.X86;
|
||||
// Process specific
|
||||
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Pointers.RemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit);
|
||||
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Pointers.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Pointers.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Pointers.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
|
||||
|
||||
// Thread Specific
|
||||
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
|
||||
}
|
||||
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, CodeGenerator(0, this, nullptr) // this is not used here
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -358,98 +358,58 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
if (!CompileThread) {
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
{
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
ThreadSharedData.OverflowExceptionInstructionAddress = Dispatcher->OverflowExceptionInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
X86JITCore::~X86JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
|
||||
}
|
||||
|
||||
void X86JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::X86JITCore::";
|
||||
for (char c : JITString) {
|
||||
db(c);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
void X86JITCore::ClearCache() {
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(CTX, CurrentCodeBuffer->Size);
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(CTX, X86JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
setNewBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
setNewBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
LOGMAN_THROW_AA_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
}
|
||||
@@ -611,42 +571,30 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((getSize() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
void *GuestEntry = getCurr<void*>();
|
||||
GuestEntry = getCurr<uint8_t*>();
|
||||
CursorEntry = getSize();
|
||||
this->IR = IR;
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(RunBlock);
|
||||
// Else we need to pause now
|
||||
mov(rax, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
|
||||
jmp(rax);
|
||||
ud2();
|
||||
|
||||
L(RunBlock);
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
setSize(getSize() + GDBSize);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(RAData != nullptr, "Needs RA");
|
||||
LOGMAN_THROW_AA_FMT(RAData != nullptr, "Needs RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -701,12 +649,13 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
{
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = getCurr<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
|
||||
@@ -761,6 +710,13 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(getCurr<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
@@ -777,32 +733,22 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(GuestExit) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
|
||||
record[0] = HostCode;
|
||||
return HostCode;
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
X86JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
}
|
||||
+84
-50
@@ -6,8 +6,10 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
@@ -24,13 +26,6 @@ using namespace Xbyak;
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -57,9 +52,7 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
|
||||
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
CodeBuffer Buffer,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~X86JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -67,7 +60,7 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -75,20 +68,75 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
}
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
void LoadConstantWithPadding(Xbyak::Reg Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Label Offset;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(Xbyak::Reg Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(Xbyak::Reg Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint64_t CursorEntry{};
|
||||
/** @} */
|
||||
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
uint64_t Entry;
|
||||
|
||||
std::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
@@ -133,6 +181,10 @@ private:
|
||||
[[nodiscard]] Xbyak::Xmm GetSrc(IR::NodeID Node) const;
|
||||
[[nodiscard]] Xbyak::Xmm GetDst(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] static Xbyak::Ymm ToYMM(const Xbyak::Xmm& xmm) {
|
||||
return Xbyak::Ymm{xmm.getIdx()};
|
||||
}
|
||||
|
||||
[[nodiscard]] Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale) const;
|
||||
|
||||
@@ -141,41 +193,23 @@ private:
|
||||
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
FEXCore::IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
bool GetSamplingData {true};
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
static uint64_t ExitFunctionLink(X86JITCore* code, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
using JCC = void (X86JITCore::*)(const Label& label, LabelType type);
|
||||
@@ -274,8 +308,6 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -284,7 +316,7 @@ private:
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -319,7 +351,7 @@ private:
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -329,6 +361,7 @@ private:
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -434,6 +467,7 @@ private:
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|x86-64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
+40
-53
@@ -7,6 +7,7 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -20,6 +21,12 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, getCurr<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
@@ -38,56 +45,30 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
ud2();
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.OverflowExceptionHandler)]);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadStopHandler)]);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_Interrupt3: // INT3
|
||||
{
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// This jump target needs to be a constant offset here
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadPauseHandler)]);
|
||||
}
|
||||
else {
|
||||
// If we don't have a gdb server attached then....crash?
|
||||
// Treat this case like HLT
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)], Op->Reason.Signal);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], Op->Reason.TrapNumber);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], Op->Reason.ErrorRegister);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], Op->Reason.si_code);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadStopHandler)]);
|
||||
}
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)]);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)]);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)]);
|
||||
break;
|
||||
default:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)]);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.UnimplementedInstructionHandler)]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -103,7 +84,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrc<RA_32>(Op->RoundMode.ID());
|
||||
|
||||
// Load old mxcsr
|
||||
// Only stores to memory
|
||||
@@ -128,15 +109,15 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushRegs();
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.PrintValue)]);
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Value.ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)]);
|
||||
}
|
||||
else {
|
||||
pextrq(rdi, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
pextrq(rdi, GetSrc(Op->Value.ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Value.ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.PrintVectorValue)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)]);
|
||||
}
|
||||
|
||||
PopRegs();
|
||||
@@ -166,6 +147,10 @@ DEF_OP(RDRAND) {
|
||||
setc(Dst.second.cvt8());
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
pause();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -174,6 +159,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -184,6 +170,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,13 +20,13 @@ DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<Xbyak::Reg, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetDst<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<Xbyak::Reg, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetDst<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
@@ -45,15 +45,15 @@ DEF_OP(CreateElementPair) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetSrc<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Upper.ID());
|
||||
RegTmp = eax;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetSrc<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Upper.ID());
|
||||
RegTmp = rax;
|
||||
break;
|
||||
}
|
||||
@@ -75,7 +75,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+347
-313
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,139 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|x86-64
|
||||
desc: relocation logic of the x86-64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
uint64_t X86JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
|
||||
}
|
||||
|
||||
void X86JITCore::LoadConstantWithPadding(Xbyak::Reg Reg, uint64_t Constant) {
|
||||
// The maximum size a move constant can be in bytes
|
||||
// Need to NOP pad to this size to ensure backpatching is always the same size
|
||||
// Calculated as:
|
||||
// [Rex]
|
||||
// [Mov op]
|
||||
// [8 byte constant]
|
||||
//
|
||||
// All other move types are smaller than this. xbyak will use a NOP slide which is quite quick
|
||||
constexpr static size_t MAX_MOVE_SIZE = 10;
|
||||
auto StartingOffset = getSize();
|
||||
mov(Reg, Constant);
|
||||
auto MoveSize = getSize() - StartingOffset;
|
||||
auto NOPPadSize = MAX_MOVE_SIZE - MoveSize;
|
||||
nop(NOPPadSize);
|
||||
}
|
||||
|
||||
X86JITCore::NamedSymbolLiteralPair X86JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
NamedSymbolLiteralPair Lit {
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void X86JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = getSize();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CursorEntry;
|
||||
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Lit.MoveABI.NamedSymbolLiteral.Symbol);
|
||||
|
||||
L(Lit.Offset);
|
||||
dq(Pointer);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
|
||||
void X86JITCore::InsertGuestRIPMove(Xbyak::Reg Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = getSize();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CursorEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.getIdx();
|
||||
|
||||
if (CTX->Config.CacheObjectCodeCompilation()) {
|
||||
LoadConstantWithPadding(Reg, Constant);
|
||||
}
|
||||
else {
|
||||
mov(Reg, Constant);
|
||||
}
|
||||
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool X86JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Place the pointer
|
||||
dq(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(CTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstantWithPadding(Xbyak::Reg64(Reloc->NamedThunkMove.RegisterIndex), Pointer);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstantWithPadding(Xbyak::Reg64(Reloc->GuestRIPMove.RegisterIndex), Pointer);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
+5
-10
@@ -37,11 +37,11 @@ LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
LOGMAN_THROW_AA_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
LOGMAN_THROW_AA_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
@@ -52,15 +52,8 @@ LookupCache::~LookupCache() {
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
|
||||
}
|
||||
|
||||
void LookupCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
// Tell the kernel we will definitely need [Address, Address+Size) mapped for the page pointer
|
||||
// Page Pointer is allocated per page, so shift by page size
|
||||
Address >>= 12;
|
||||
Size >>= 12;
|
||||
madvise(reinterpret_cast<void*>(PagePointer + Address), Size, MADV_WILLNEED);
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8, MADV_DONTNEED);
|
||||
madvise(reinterpret_cast<void*>(PageMemory), CODE_SIZE, MADV_DONTNEED);
|
||||
@@ -68,6 +61,8 @@ void LookupCache::ClearL2Cache() {
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1
|
||||
madvise(reinterpret_cast<void*>(L1Pointer), L1_SIZE, MADV_DONTNEED);
|
||||
// Clear L2
|
||||
|
||||
+76
-56
@@ -7,6 +7,7 @@
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -24,38 +25,73 @@ public:
|
||||
LookupCache(FEXCore::Context::Context *CTX);
|
||||
~LookupCache();
|
||||
|
||||
using LookupCacheIter = uintptr_t;
|
||||
uintptr_t End() { return 0; }
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
auto HostCode = FindCodePointerForAddress(Address);
|
||||
if (HostCode) {
|
||||
return HostCode;
|
||||
} else {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
// Try L1, no lock needed
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
} else {
|
||||
return 0;
|
||||
// L2 and L3 need to be locked
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize -1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address)
|
||||
{
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto InsertPoint =
|
||||
#endif
|
||||
BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LOGMAN_THROW_A_FMT(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
CodePages[CurrentPage].push_back(Address);
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto &CodePage = CodePages[CurrentPage];
|
||||
rv |= CodePage.size() == 0;
|
||||
CodePage.push_back(Address);
|
||||
}
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -65,6 +101,8 @@ public:
|
||||
|
||||
void Erase(uint64_t Address) {
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks.lower_bound({Address, 0});
|
||||
auto upper = BlockLinks.upper_bound({Address, UINTPTR_MAX});
|
||||
@@ -78,7 +116,10 @@ public:
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = L1Entry.HostCode = 0;
|
||||
L1Entry.GuestCode = 0;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
// Do full map
|
||||
@@ -101,14 +142,14 @@ public:
|
||||
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks.insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
|
||||
void HintUsedRange(uint64_t Address, uint64_t Size);
|
||||
|
||||
uintptr_t GetL1Pointer() const { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() const { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
@@ -116,8 +157,19 @@ public:
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
@@ -167,38 +219,6 @@ private:
|
||||
return PageMemory + NewBase;
|
||||
}
|
||||
|
||||
uintptr_t FindCodePointerForAddress(uint64_t Address) {
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize -1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == FullAddress)
|
||||
{
|
||||
L1Entry.GuestCode = FullAddress;
|
||||
return L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
}
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
|
||||
+84
@@ -0,0 +1,84 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie{};
|
||||
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock{};
|
||||
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
bool MultiBlock : 1;
|
||||
|
||||
// TSO enabled
|
||||
bool TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
bool ABILocalFlags : 1;
|
||||
|
||||
// ABI no PF unsafe optimization
|
||||
bool ABINoPF : 1;
|
||||
|
||||
// Static register allocation enabled
|
||||
bool SRA : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
bool ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
bool Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
bool x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
SRA == other.SRA &&
|
||||
ParanoidTSO == other.ParanoidTSO &&
|
||||
Is64BitMode == other.Is64BitMode &&
|
||||
SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(CodeObjectSerializationConfig const &other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash{};
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
Hash <<= 1; Hash |= other.SRA;
|
||||
Hash <<= 1; Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1; Hash |= other.Is64BitMode;
|
||||
Hash <<= 2; Hash |= other.SMCChecks;
|
||||
Hash <<= 1; Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <sys/uio.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
auto BaseFilename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = std::make_unique<CodeRegionEntry>(
|
||||
Base,
|
||||
Size,
|
||||
Offset,
|
||||
filename,
|
||||
NamedRegionHandler->DefaultCodeHeader(Base, Offset)
|
||||
);
|
||||
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
std::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
}
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::Context *ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ABINoPF = ctx->Config.ABINoPF;
|
||||
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion *>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion *>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService *This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler { &NamedRegionHandler , this }
|
||||
, NamedRegionHandler { ctx } {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
WorkerThreadShuttingDown = true;
|
||||
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, std::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
CodeObjectFileSection const *CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16] = "ObjectCodeSeri\0";
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto &it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,459 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
#include "Interface/Core/ObjectCache/CodeObjectSerializationConfig.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <shared_mutex>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// XXX: Does this need to be signal safe?
|
||||
using CodeSerializationMutex = std::shared_mutex;
|
||||
struct CodeSerializationData {
|
||||
};
|
||||
|
||||
struct CodeObjectFileSection {
|
||||
bool Serialized;
|
||||
bool Invalid;
|
||||
const CodeSerializationData *Data;
|
||||
const char *HostCode;
|
||||
uint64_t NumRelocations;
|
||||
const char *Relocations;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is the file header that lives at the start of an object cache file
|
||||
*
|
||||
* This header is updated from multiple processes!
|
||||
* Care must be taken to use OS locks when updating the file backing including this header
|
||||
*/
|
||||
struct CodeObjectSerializationHeader {
|
||||
// The configuration that this file has
|
||||
CodeObjectSerializationConfig Config;
|
||||
// The original RIP that this object section was mapped at
|
||||
uint64_t OriginalBase{};
|
||||
// The original offset in to the file that this object section was loaded from
|
||||
uint64_t OriginalOffset{};
|
||||
// Total amount of code that should be in this file
|
||||
uint64_t TotalCodeSize{};
|
||||
// Used to reserve the TSL map
|
||||
uint64_t NumCodeEntries{};
|
||||
// The number of relocations that point to this section
|
||||
uint64_t NumRelocationsTo{};
|
||||
// Total relocations in this file
|
||||
uint64_t TotalRelocationsCount{};
|
||||
};
|
||||
|
||||
struct CodeRegionEntry {
|
||||
/**
|
||||
* @name Threaded initialization objects for the initial object creation
|
||||
* @{ */
|
||||
// Base address in memory where the code region is at
|
||||
uint64_t Base{};
|
||||
|
||||
// Size of this code entry
|
||||
uint64_t Size{};
|
||||
|
||||
// The offset inside the file that is mapped to Base
|
||||
uint64_t Offset{};
|
||||
|
||||
// Filename of the object
|
||||
std::string Filename{};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader{};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
std::string ObjectEntrySourceFilename{};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
bool StillSerializing {true};
|
||||
|
||||
// Long lived FD for serialization if we have multiple jobs to serialize
|
||||
// Bursts of code entries are common and this reduces file lock overhead
|
||||
//
|
||||
// Especially useful over network mounts where file locks are very slow
|
||||
int CurrentSerializedFD {-1};
|
||||
|
||||
/**
|
||||
* @name Objects required to sync objects between threads
|
||||
* @{ */
|
||||
// Refcount for the number of outstanding code entries waiting to be written for this object section
|
||||
CodeSerializationMutex ObjectJobRefCountMutex;
|
||||
|
||||
// Refcount for outstanding named object region entry loading itself
|
||||
// Will block JIT code cache look up when this has a unique_lock held
|
||||
CodeSerializationMutex NamedJobRefCountMutex;
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Object Entry data management
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @name This is the raw file data that we loaded from the code region entry file
|
||||
* @{ */
|
||||
char *CodeData{};
|
||||
size_t FileSize{};
|
||||
|
||||
std::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
tsl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
CodeRegionEntry() = default;
|
||||
|
||||
// Initializer specifically for threaded loading
|
||||
CodeRegionEntry(uint64_t Base,
|
||||
uint64_t Size,
|
||||
uint64_t Offset,
|
||||
std::string const &Filename,
|
||||
CodeObjectSerializationHeader const &DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
, Offset {Offset}
|
||||
, Filename {Filename}
|
||||
, EntryHeader {DefaultHeader} {
|
||||
}
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = std::map<uint64_t, std::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = std::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
|
||||
class AsyncJobHandler final {
|
||||
public:
|
||||
/**
|
||||
* @brief Structure containing all the data required to async serialize code objects
|
||||
*/
|
||||
struct SerializationJobData {
|
||||
uint64_t GuestRIP; ///< The RIP for the guest
|
||||
// XXX: Support multiblock
|
||||
uint64_t GuestCodeLength; ///< The Guest's code length
|
||||
uint64_t GuestCodeHash; ///< Hash of the guest code
|
||||
|
||||
void *HostCodeBegin; ///< Host JIT code starting memory address
|
||||
size_t HostCodeLength; ///< Host JIT code length
|
||||
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
|
||||
|
||||
// This is the thread specific ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
|
||||
// This way it will wait until the async job handler is complete with it.
|
||||
CodeSerializationMutex *ThreadJobRefCount;
|
||||
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
* @{ */
|
||||
// This is the code region's ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
|
||||
CodeSerializationMutex *ObjectJobRefCountMutexPtr;
|
||||
|
||||
// This is the code region iterator to reduce the number of map lookups
|
||||
// This will remain valid while jobs are outstanding for this region
|
||||
CodeRegionMapType::iterator CodeRegionIterator;
|
||||
/** @} */
|
||||
};
|
||||
|
||||
AsyncJobHandler(NamedRegionObjectHandler *NamedRegionHandler, CodeObjectSerializeService *CodeObjectCacheService)
|
||||
: NamedRegionHandler {NamedRegionHandler}
|
||||
, CodeObjectCacheService {CodeObjectCacheService} {}
|
||||
|
||||
protected:
|
||||
friend class CodeObjectSerializeService;
|
||||
friend class NamedRegionObjectHandler;
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Async named region handling
|
||||
* @{ */
|
||||
/**
|
||||
* @brief The async named region jobs to handle.
|
||||
*
|
||||
* Only two, Code serialization goes in to a different queue.
|
||||
*/
|
||||
enum class NamedRegionJobType {
|
||||
JOB_ADD_NAMED_REGION,
|
||||
JOB_REMOVE_NAMED_REGION,
|
||||
};
|
||||
|
||||
class NamedRegionWorkItem {
|
||||
public:
|
||||
NamedRegionJobType GetType() const { return Type; }
|
||||
|
||||
protected:
|
||||
friend class WorkItemAddNamedRegion;
|
||||
NamedRegionWorkItem(NamedRegionJobType type)
|
||||
: Type {type} {}
|
||||
|
||||
private:
|
||||
NamedRegionJobType Type;
|
||||
};
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const std::string &base, const std::string &filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry}
|
||||
{}
|
||||
const std::string BaseFilename;
|
||||
const std::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, std::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
, Entry {std::move(entry)} {}
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
std::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
private:
|
||||
NamedRegionObjectHandler *NamedRegionHandler;
|
||||
CodeObjectSerializeService *CodeObjectCacheService;
|
||||
};
|
||||
|
||||
class NamedRegionObjectHandler final {
|
||||
public:
|
||||
NamedRegionObjectHandler(FEXCore::Context::Context *ctx);
|
||||
|
||||
void HandleNamedRegionObjectJobs();
|
||||
|
||||
CodeObjectSerializationConfig const &GetDefaultSerializationConfig() const {
|
||||
return DefaultSerializationConfig;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
// Return a default code header based off the default serialization config
|
||||
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
|
||||
return CodeObjectSerializationHeader {
|
||||
.Config = DefaultSerializationConfig,
|
||||
.OriginalBase = Base,
|
||||
.OriginalOffset = Offset,
|
||||
.NumCodeEntries = 0,
|
||||
.NumRelocationsTo = 0,
|
||||
.TotalRelocationsCount = 0,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds an asynchronous add named region work item to the object queue
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const std::string &base, const std::string &filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(std::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
|
||||
base,
|
||||
filename,
|
||||
executable,
|
||||
entry
|
||||
));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(std::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
|
||||
Base,
|
||||
Size,
|
||||
std::move(Entry)
|
||||
));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
private:
|
||||
// Code version. If the code emission changes then this needs to increment
|
||||
constexpr static uint32_t CODE_VERSION = 0x0;
|
||||
|
||||
// Default cookie header for the file header
|
||||
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
|
||||
|
||||
// Code serialization config for our current process configuration
|
||||
CodeObjectSerializationConfig DefaultSerializationConfig;
|
||||
|
||||
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
|
||||
std::atomic<uint64_t> NamedWorkQueueJobs{};
|
||||
|
||||
// Mutex for ading new jobs to the work queue
|
||||
std::mutex NamedWorkQueueMutex{};
|
||||
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
std::queue<std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Context specific code object serialization class
|
||||
*
|
||||
* Contains everything required for FEXCore to serialize code objects
|
||||
*/
|
||||
class CodeObjectSerializeService final {
|
||||
public:
|
||||
CodeObjectSerializeService(FEXCore::Context::Context *ctx);
|
||||
|
||||
/**
|
||||
* @brief Initialize the internal interface
|
||||
*
|
||||
* Is a public interface to allow the service to reinitialize after forking
|
||||
*/
|
||||
void Initialize();
|
||||
|
||||
/**
|
||||
* @brief Safely shut down the Code Object serialization service.
|
||||
*
|
||||
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
|
||||
*/
|
||||
void Shutdown();
|
||||
|
||||
/**
|
||||
* @name Async interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Loads a named region in to the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address that this named region is loaded
|
||||
* @param Size - The size of the region
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Unloads a named region from the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address of the named region
|
||||
* @param Size - The size of the region
|
||||
*/
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds a code object serialization job. As async as possible.
|
||||
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(std::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Synchronous interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Synchronously waits for this thread's job queue to become empty.
|
||||
*
|
||||
* This is necessary for when a thread is shutting down
|
||||
*
|
||||
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
|
||||
*/
|
||||
static void WaitForEmptyJobQueue(CodeSerializationMutex *ThreadJobRefCount) {
|
||||
// Once the shared mutex is empty this unique lock will be gained
|
||||
std::unique_lock lk {*ThreadJobRefCount};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Fetches object code from the Code Object Cache for JIT.
|
||||
*
|
||||
* @param GuestRIP - Which GuestRIP to search the cache for
|
||||
*
|
||||
* @return Data required for the JIT to relocate the Object code.
|
||||
*/
|
||||
CodeObjectFileSection const *FetchCodeObjectFromCache(uint64_t GuestRIP);
|
||||
/** @} */
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
/**
|
||||
* @brief Safely closes out code object regions from the map
|
||||
*
|
||||
* @param it - iterator to do a closure on
|
||||
*/
|
||||
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it);
|
||||
|
||||
CodeSerializationMutex &GetEntryMapMutex() { return EntryMapMutex; }
|
||||
CodeSerializationMutex &GetUnrelocatedEntryMapMutex() { return EntryMapMutex; }
|
||||
|
||||
CodeRegionMapType &GetEntryMap() { return AddressToEntryMap; }
|
||||
CodeRegionPtrMapType &GetUnrelocatedEntryMap() { return UnrelocatedAddressToEntryMap; }
|
||||
|
||||
/**
|
||||
* @brief Notify the async thread that it has work to do
|
||||
*/
|
||||
void NotifyWork() { WorkAvailable.NotifyOne(); }
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
Event WorkAvailable{};
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
// Mutex to hold when modifying the entry maps
|
||||
CodeSerializationMutex EntryMapMutex;
|
||||
CodeSerializationMutex UnrelocatedEntryMapMutex;
|
||||
|
||||
// Entry maps
|
||||
CodeRegionMapType AddressToEntryMap;
|
||||
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
#pragma once
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum class RelocationTypes : uint8_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
RelocationTypes Type;
|
||||
};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
NamedSymbol Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
};
|
||||
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
};
|
||||
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
|
||||
// The unrelocated RIP that is being moved
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
};
|
||||
}
|
||||
+499
-90
@@ -17,6 +17,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -350,7 +351,7 @@ void OpDispatchBuilder::SecondaryALUOp(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", IROp);
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", ToUnderlying(IROp));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1443,6 +1444,11 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// But this would result in a zext on 64bit, which would ruin the no-op nature of the instruction
|
||||
// So x86-64 spec mandates this special case that even though it is a 32bit instruction and
|
||||
// is supposed to zext the result, it is a true no-op
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) {
|
||||
// If this instruction has a REP prefix then this is architectually defined to be a `PAUSE` instruction.
|
||||
// On older processors this ends up being a true `REP NOP` which is why they stuck this here.
|
||||
_Yield();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1584,7 +1590,12 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
case 2: // CS
|
||||
case FEXCore::X86State::REG_R9: // CS
|
||||
// CPL3 can't write to this
|
||||
_Break(FEXCore::IR::Break_InvalidInstruction, 0);
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
.si_code = 0,
|
||||
});
|
||||
break;
|
||||
case 3: // SS
|
||||
case FEXCore::X86State::REG_R10: // SS
|
||||
@@ -3353,7 +3364,9 @@ void OpDispatchBuilder::ReadSegmentReg(OpcodeArgs) {
|
||||
|
||||
template<OpDispatchBuilder::Segment Seg>
|
||||
void OpDispatchBuilder::WriteSegmentReg(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
// Documentation claims that the 32-bit version of this instruction inserts in to the lower 32-bits of the segment
|
||||
// This is incorrect and it instead zero extends the 32-bit value to 64-bit
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
if constexpr (Seg == Segment::FS) {
|
||||
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs));
|
||||
@@ -4496,7 +4509,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
[[maybe_unused]] const FEXCore::IR::IROp_Header *IROp =
|
||||
RealNode->Op(DualListData.DataBegin());
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
|
||||
// Let's walk the jump blocks and see if we have handled every block target
|
||||
for (auto &Handler : JumpTargets) {
|
||||
@@ -4522,7 +4535,7 @@ uint8_t OpDispatchBuilder::GetDstSize(X86Tables::DecodedOp Op) const {
|
||||
|
||||
const uint32_t DstSizeFlag = X86Tables::DecodeFlags::GetSizeDstFlags(Op->Flags);
|
||||
const uint8_t Size = Sizes[DstSizeFlag];
|
||||
LOGMAN_THROW_A_FMT(Size != 0, "Invalid destination size for op");
|
||||
LOGMAN_THROW_AA_FMT(Size != 0, "Invalid destination size for op");
|
||||
return Size;
|
||||
}
|
||||
|
||||
@@ -4540,7 +4553,7 @@ uint8_t OpDispatchBuilder::GetSrcSize(X86Tables::DecodedOp Op) const {
|
||||
|
||||
const uint32_t SrcSizeFlag = X86Tables::DecodeFlags::GetSizeSrcFlags(Op->Flags);
|
||||
const uint8_t Size = Sizes[SrcSizeFlag];
|
||||
LOGMAN_THROW_A_FMT(Size != 0, "Invalid destination size for op");
|
||||
LOGMAN_THROW_AA_FMT(Size != 0, "Invalid destination size for op");
|
||||
return Size;
|
||||
}
|
||||
|
||||
@@ -4604,7 +4617,7 @@ OrderedNode *OpDispatchBuilder::AppendSegmentOffset(OrderedNode *Value, uint32_t
|
||||
return Value;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad) {
|
||||
OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad, MemoryAccessType AccessType) {
|
||||
LOGMAN_THROW_A_FMT(Operand.IsGPR() ||
|
||||
Operand.IsLiteral() ||
|
||||
Operand.IsGPRDirect() ||
|
||||
@@ -4615,7 +4628,6 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
|
||||
OrderedNode *Src {nullptr};
|
||||
bool LoadableType = false;
|
||||
bool StackAccess = false;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const uint32_t AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
@@ -4635,7 +4647,14 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
}
|
||||
else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][Operand.Data.GPR.HighBits ? 1 : 0]));
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(Core::CPUState, xmm.avx.data[gprIndex][highIndex]));
|
||||
} else {
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(Core::CPUState, xmm.sse.data[gprIndex][highIndex]));
|
||||
}
|
||||
}
|
||||
else {
|
||||
Src = _LoadContext(OpSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[gpr]) + (Operand.Data.GPR.HighBits ? 1 : 0));
|
||||
@@ -4644,7 +4663,9 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
else if (Operand.IsGPRDirect()) {
|
||||
Src = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPR.GPR]));
|
||||
LoadableType = true;
|
||||
StackAccess = Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsGPRIndirect()) {
|
||||
auto GPR = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPRIndirect.GPR]));
|
||||
@@ -4653,7 +4674,9 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
Src = _Add(GPR, Constant);
|
||||
|
||||
LoadableType = true;
|
||||
StackAccess = Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsRIPRelative()) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
@@ -4675,7 +4698,9 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
auto Constant = _Constant(GPRSize * 8, Operand.Data.SIB.Scale);
|
||||
Tmp = _Mul(Tmp, Constant);
|
||||
}
|
||||
StackAccess |= Operand.Data.SIB.Index == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.SIB.Index == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
|
||||
if (Operand.Data.SIB.Base != FEXCore::X86State::REG_INVALID) {
|
||||
@@ -4687,7 +4712,10 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
else {
|
||||
Tmp = GPR;
|
||||
}
|
||||
StackAccess |= Operand.Data.SIB.Base == FEXCore::X86State::REG_RSP;
|
||||
|
||||
if (Operand.Data.SIB.Base == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
|
||||
if (Operand.Data.SIB.Offset) {
|
||||
@@ -4722,7 +4750,7 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
if ((LoadableType && LoadData) || ForceLoad) {
|
||||
Src = AppendSegmentOffset(Src, Flags);
|
||||
|
||||
if (StackAccess) {
|
||||
if (AccessType == MemoryAccessType::ACCESS_NONTSO || AccessType == MemoryAccessType::ACCESS_STREAM) {
|
||||
Src = _LoadMem(Class, OpSize, Src, Align == -1 ? OpSize : Align);
|
||||
}
|
||||
else {
|
||||
@@ -4737,12 +4765,12 @@ OrderedNode *OpDispatchBuilder::GetRelocatedPC(FEXCore::X86Tables::DecodedOp con
|
||||
return _EntrypointOffset(Op->PC + Op->InstSize + Offset - Entry, GPRSize);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad) {
|
||||
OrderedNode *OpDispatchBuilder::LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad, MemoryAccessType AccessType) {
|
||||
const uint8_t OpSize = GetSrcSize(Op);
|
||||
return LoadSource_WithOpSize(Class, Op, Operand, OpSize, Flags, Align, LoadData, ForceLoad);
|
||||
return LoadSource_WithOpSize(Class, Op, Operand, OpSize, Flags, Align, LoadData, ForceLoad, AccessType);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align) {
|
||||
void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType) {
|
||||
LOGMAN_THROW_A_FMT(Operand.IsGPR() ||
|
||||
Operand.IsLiteral() ||
|
||||
Operand.IsGPRDirect() ||
|
||||
@@ -4755,7 +4783,6 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
// 32bit ops ZEXT the result to 64bit
|
||||
OrderedNode *MemStoreDst {nullptr};
|
||||
bool MemStore = false;
|
||||
bool StackAccess = false;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const uint32_t AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
@@ -4769,7 +4796,14 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
_StoreContext(OpSize, Class, Src, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
}
|
||||
else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][Operand.Data.GPR.HighBits ? 1 : 0]));
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(Core::CPUState, xmm.avx.data[gprIndex][highIndex]));
|
||||
} else {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(Core::CPUState, xmm.sse.data[gprIndex][highIndex]));
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (GPRSize == 8 && OpSize == 4) {
|
||||
@@ -4777,11 +4811,11 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
// For all other sizes, the upper bits are guaranteed to already be zero
|
||||
OrderedNode *Value = GetOpSize(Src) == 8 ? _Bfe(4, 32, 0, Src) : Src;
|
||||
|
||||
LOGMAN_THROW_A_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
LOGMAN_THROW_AA_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
_StoreContext(GPRSize, Class, Value, offsetof(FEXCore::Core::CPUState, gregs[gpr]));
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(!(GPRSize == 4 && OpSize > 4), "Oops had a {} GPR load", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(!(GPRSize == 4 && OpSize > 4), "Oops had a {} GPR load", OpSize);
|
||||
_StoreContext(std::min(GPRSize, OpSize), Class, Src, offsetof(FEXCore::Core::CPUState, gregs[gpr]) + (Operand.Data.GPR.HighBits ? 1 : 0));
|
||||
}
|
||||
}
|
||||
@@ -4789,7 +4823,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
else if (Operand.IsGPRDirect()) {
|
||||
MemStoreDst = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPR.GPR]));
|
||||
MemStore = true;
|
||||
StackAccess = Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsGPRIndirect()) {
|
||||
auto GPR = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPRIndirect.GPR]));
|
||||
@@ -4797,7 +4833,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
|
||||
MemStoreDst = _Add(GPR, Constant);
|
||||
MemStore = true;
|
||||
StackAccess = Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsRIPRelative()) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
@@ -4867,7 +4905,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
auto DestAddr = _Add(MemStoreDst, _Constant(8));
|
||||
_StoreMem(GPRClass, 2, DestAddr, Upper, std::min<uint8_t>(Align, 8));
|
||||
} else {
|
||||
if (StackAccess) {
|
||||
if (AccessType == MemoryAccessType::ACCESS_NONTSO || AccessType == MemoryAccessType::ACCESS_STREAM) {
|
||||
_StoreMem(Class, OpSize, MemStoreDst, Src, Align == -1 ? OpSize : Align);
|
||||
}
|
||||
else {
|
||||
@@ -4877,17 +4915,23 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align) {
|
||||
StoreResult_WithOpSize(Class, Op, Operand, Src, GetDstSize(Op), Align);
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType) {
|
||||
StoreResult_WithOpSize(Class, Op, Operand, Src, GetDstSize(Op), Align, AccessType);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align) {
|
||||
StoreResult(Class, Op, Op->Dest, Src, Align);
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType) {
|
||||
StoreResult(Class, Op, Op->Dest, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx} {
|
||||
: IREmitter {ctx->OpDispatcherAllocator}
|
||||
, CTX {ctx} {
|
||||
ResetWorkingList();
|
||||
InstallHostSpecificOpcodeHandlers();
|
||||
}
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
: IREmitter {Allocator}
|
||||
, CTX {nullptr} {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ResetWorkingList() {
|
||||
@@ -4909,6 +4953,11 @@ void OpDispatchBuilder::MOVGPROp(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, 1);
|
||||
StoreResult(GPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
bool RequiresMask = false;
|
||||
FEXCore::IR::IROps IROp;
|
||||
@@ -5000,7 +5049,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", IROp);
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", ToUnderlying(IROp));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -5040,37 +5089,58 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
FEXCore::IR::BreakReason Reason{};
|
||||
uint8_t Literal{};
|
||||
bool setRIP = false;
|
||||
IR::BreakDefinition Reason;
|
||||
bool SetRIPToNext = false;
|
||||
|
||||
switch (Op->OP) {
|
||||
case 0xCD:
|
||||
Reason = FEXCore::IR::Break_Interrupt;
|
||||
Literal = Op->Src[0].Data.Literal.Value;
|
||||
case 0xCD: { // INT imm8
|
||||
uint8_t Literal = Op->Src[0].Data.Literal.Value;
|
||||
|
||||
if (Literal == 0x80) {
|
||||
// Syscall on linux
|
||||
SyscallOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Reason.ErrorRegister = Literal << 3 | (0b010);
|
||||
Reason.Signal = SIGSEGV;
|
||||
// GP is raised when task-gate isn't setup to be valid
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
}
|
||||
case 0xCE: // INTO
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_OF;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
case 0xF1: // INT1
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_DB;
|
||||
Reason.si_code = 1;
|
||||
SetRIPToNext = true;
|
||||
break;
|
||||
case 0xCE:
|
||||
Reason = FEXCore::IR::Break_Overflow;
|
||||
break;
|
||||
case 0xF1:
|
||||
Reason = FEXCore::IR::Break_Interrupt;
|
||||
break;
|
||||
case 0xF4: {
|
||||
Reason = FEXCore::IR::Break_Halt;
|
||||
setRIP = true;
|
||||
case 0xF4: { // HLT
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
}
|
||||
case 0x0B:
|
||||
Reason = FEXCore::IR::Break_Interrupt;
|
||||
case 0x0B: // UD2
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGILL;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_UD;
|
||||
Reason.si_code = 2;
|
||||
break;
|
||||
case 0xCC:
|
||||
Reason = FEXCore::IR::Break_Interrupt3;
|
||||
setRIP = true;
|
||||
case 0xCC: // INT3
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_BP;
|
||||
Reason.si_code = 0x80;
|
||||
SetRIPToNext = true;
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -5079,13 +5149,17 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
if (setRIP) {
|
||||
BlockSetRIP = setRIP;
|
||||
if (SetRIPToNext) {
|
||||
BlockSetRIP = SetRIPToNext;
|
||||
|
||||
// We want to set RIP to the next instruction after HLT/INT3
|
||||
// We want to set RIP to the next instruction after INT3/INT1
|
||||
auto NewRIP = GetRelocatedPC(Op);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
}
|
||||
else if (Op->OP != 0xCE) {
|
||||
auto NewRIP = GetRelocatedPC(Op, -Op->InstSize);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
}
|
||||
|
||||
if (Op->OP == 0xCE) { // Conditional to only break if Overflow == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
@@ -5098,7 +5172,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(Reason, Literal);
|
||||
_Break(Reason);
|
||||
|
||||
// Make sure to start a new block after ending this one
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(FalseBlock);
|
||||
@@ -5106,7 +5180,8 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
}
|
||||
else {
|
||||
_Break(Reason, Literal);
|
||||
BlockSetRIP = true;
|
||||
_Break(Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5209,7 +5284,13 @@ void OpDispatchBuilder::UnimplementedOp(OpcodeArgs) {
|
||||
// We don't actually support this instruction
|
||||
// Multiblock may hit it though
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(FEXCore::IR::Break_Unimplemented, 0);
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
.si_code = 0,
|
||||
});
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
if (Multiblock) {
|
||||
@@ -5227,12 +5308,114 @@ void OpDispatchBuilder::InvalidOp(OpcodeArgs) {
|
||||
// We don't actually support this instruction
|
||||
// Multiblock may hit it though
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(FEXCore::IR::Break_InvalidInstruction, 0);
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
.si_code = 0,
|
||||
});
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
|
||||
void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
static bool Initialized = false;
|
||||
if (!CTX || Initialized) {
|
||||
// IRCompaction doesn't set a CTX and doesn't need this anyway
|
||||
return;
|
||||
}
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F38_SHA[] = {
|
||||
{OPD(PF_38_NONE, 0xC8), 1, &OpDispatchBuilder::SHA1NEXTEOp},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, &OpDispatchBuilder::SHA1MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, &OpDispatchBuilder::SHA1MSG2Op},
|
||||
{OPD(PF_38_NONE, 0xCB), 1, &OpDispatchBuilder::SHA256RNDS2Op},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F38_AES[] = {
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
|
||||
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
|
||||
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
|
||||
{OPD(PF_38_66, 0xDF), 1, &OpDispatchBuilder::AESDecLastOp},
|
||||
};
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F38_CRC[] = {
|
||||
{OPD(PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3A_AES[] = {
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
};
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
#undef OPD
|
||||
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
constexpr uint16_t PF_NONE = 0;
|
||||
constexpr uint16_t PF_66 = 2;
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryExtensionOp_RDRAND[] = {
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryModRMExtensionOp_CLZero[] = {
|
||||
{((3 << 3) | 4), 1, &OpDispatchBuilder::CLZeroOp},
|
||||
};
|
||||
|
||||
auto InstallToTable = [](auto& FinalTable, auto& LocalTable) {
|
||||
for (auto Op : LocalTable) {
|
||||
auto OpNum = std::get<0>(Op);
|
||||
auto Dispatcher = std::get<2>(Op);
|
||||
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
|
||||
LOGMAN_THROW_A_FMT(FinalTable[OpNum + i].OpcodeDispatcher == nullptr, "Duplicate Entry");
|
||||
FinalTable[OpNum + i].OpcodeDispatcher = Dispatcher;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (CTX->HostFeatures.SupportsCRC) {
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38_CRC);
|
||||
}
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38_SHA);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAES) {
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38_AES);
|
||||
InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3A_AES);
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
InstallToTable(FEXCore::X86Tables::SecondModRMTableOps, SecondaryModRMExtensionOp_CLZero);
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsRAND) {
|
||||
InstallToTable(FEXCore::X86Tables::SecondInstGroupOps, SecondaryExtensionOp_RDRAND);
|
||||
}
|
||||
Initialized = true;
|
||||
}
|
||||
|
||||
void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
constexpr std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> BaseOpTable[] = {
|
||||
// Instructions
|
||||
@@ -5361,7 +5544,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xBD, 1, &OpDispatchBuilder::BSROp}, // BSF
|
||||
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
|
||||
{0xC0, 2, &OpDispatchBuilder::XADDOp},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPROp<0>},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::PExtrOp<2>},
|
||||
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
|
||||
@@ -5375,7 +5558,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x17, 1, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
|
||||
@@ -5437,7 +5620,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xE3, 1, &OpDispatchBuilder::PAVGOp<2>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::PSUBSOp<1, true>},
|
||||
{0xE9, 1, &OpDispatchBuilder::PSUBSOp<2, true>},
|
||||
{0xEA, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMIN, 2>},
|
||||
@@ -5665,7 +5848,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<8>},
|
||||
@@ -5739,7 +5922,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorOp},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQSUB, 1>},
|
||||
{0xE9, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQSUB, 2>},
|
||||
{0xEA, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMIN, 2>},
|
||||
@@ -5794,15 +5977,8 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
// GROUP 12
|
||||
@@ -5841,10 +6017,6 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::FenceOp<FEXCore::IR::Fence_LoadStore.Val>}, //MFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, //SFENCE
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
@@ -5860,6 +6032,15 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_66, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryExtensionOpTable_64[] = {
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryModRMExtensionOpTable[] = {
|
||||
@@ -5868,13 +6049,242 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
{((3 << 3) | 4), 1, &OpDispatchBuilder::CLZeroOp},
|
||||
|
||||
};
|
||||
// Top bit indicating if it needs to be repeated with {0x40, 0x80} or'd in
|
||||
// All OPDReg versions need it
|
||||
#define OPDReg(op, reg) ((1 << 15) | ((op - 0xD8) << 8) | (reg << 3))
|
||||
#define OPD(op, modrmop) (((op - 0xD8) << 8) | modrmop)
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> X87F64OpTable[] = {
|
||||
{OPDReg(0xD8, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xD8, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xD8, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xD8, 0xC0), 8, &OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xC8), 8, &OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xD0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
{OPD(0xD8, 0xD8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
{OPD(0xD8, 0xE0), 8, &OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xE8), 8, &OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xF0), 8, &OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xF8), 8, &OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD9, 0) | 0x00, 8, &OpDispatchBuilder::FLDF64<32>},
|
||||
|
||||
// 1 = Invalid
|
||||
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::FSTF64<32>},
|
||||
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::FSTF64<32>},
|
||||
|
||||
{OPDReg(0xD9, 4) | 0x00, 8, &OpDispatchBuilder::X87LDENVF64},
|
||||
|
||||
{OPDReg(0xD9, 5) | 0x00, 8, &OpDispatchBuilder::X87FLDCWF64},
|
||||
|
||||
{OPDReg(0xD9, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSTENV},
|
||||
|
||||
{OPDReg(0xD9, 7) | 0x00, 8, &OpDispatchBuilder::X87FSTCW},
|
||||
|
||||
{OPD(0xD9, 0xC0), 8, &OpDispatchBuilder::FLDF64<80>},
|
||||
{OPD(0xD9, 0xC8), 8, &OpDispatchBuilder::FXCH},
|
||||
{OPD(0xD9, 0xD0), 1, &OpDispatchBuilder::NOPOp}, // FNOP
|
||||
// D1 = Invalid
|
||||
// D8 = Invalid
|
||||
{OPD(0xD9, 0xE0), 1, &OpDispatchBuilder::FCHSF64},
|
||||
{OPD(0xD9, 0xE1), 1, &OpDispatchBuilder::FABSF64},
|
||||
// E2 = Invalid
|
||||
{OPD(0xD9, 0xE4), 1, &OpDispatchBuilder::FTSTF64},
|
||||
{OPD(0xD9, 0xE5), 1, &OpDispatchBuilder::X87FXAMF64},
|
||||
// E6 = Invalid
|
||||
{OPD(0xD9, 0xE8), 1, &OpDispatchBuilder::FLDF64_Const<0x3FF0000000000000>}, // 1.0
|
||||
{OPD(0xD9, 0xE9), 1, &OpDispatchBuilder::FLDF64_Const<0x400A934F0979A372>}, // log2l(10)
|
||||
{OPD(0xD9, 0xEA), 1, &OpDispatchBuilder::FLDF64_Const<0x3FF71547652B82FE>}, // log2l(e)
|
||||
{OPD(0xD9, 0xEB), 1, &OpDispatchBuilder::FLDF64_Const<0x400921FB54442D18>}, // pi
|
||||
{OPD(0xD9, 0xEC), 1, &OpDispatchBuilder::FLDF64_Const<0x3FD34413509F79FF>}, // log10l(2)
|
||||
{OPD(0xD9, 0xED), 1, &OpDispatchBuilder::FLDF64_Const<0x3FE62E42FEFA39EF>}, // log(2)
|
||||
{OPD(0xD9, 0xEE), 1, &OpDispatchBuilder::FLDF64_Const<0>}, // 0.0
|
||||
|
||||
// EF = Invalid
|
||||
{OPD(0xD9, 0xF0), 1, &OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64F2XM1>},
|
||||
{OPD(0xD9, 0xF1), 1, &OpDispatchBuilder::X87FYL2XF64},
|
||||
{OPD(0xD9, 0xF2), 1, &OpDispatchBuilder::X87TANF64},
|
||||
{OPD(0xD9, 0xF3), 1, &OpDispatchBuilder::X87ATANF64},
|
||||
{OPD(0xD9, 0xF4), 1, &OpDispatchBuilder::FXTRACTF64},
|
||||
{OPD(0xD9, 0xF5), 1, &OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM1>},
|
||||
{OPD(0xD9, 0xF6), 1, &OpDispatchBuilder::X87ModifySTP<false>},
|
||||
{OPD(0xD9, 0xF7), 1, &OpDispatchBuilder::X87ModifySTP<true>},
|
||||
{OPD(0xD9, 0xF8), 1, &OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM>},
|
||||
{OPD(0xD9, 0xF9), 1, &OpDispatchBuilder::X87FYL2XF64},
|
||||
{OPD(0xD9, 0xFA), 1, &OpDispatchBuilder::FSQRTF64},
|
||||
{OPD(0xD9, 0xFB), 1, &OpDispatchBuilder::X87SinCosF64},
|
||||
{OPD(0xD9, 0xFC), 1, &OpDispatchBuilder::FRNDINTF64},
|
||||
{OPD(0xD9, 0xFD), 1, &OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64SCALE>},
|
||||
{OPD(0xD9, 0xFE), 1, &OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64SIN>},
|
||||
{OPD(0xD9, 0xFF), 1, &OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64COS>},
|
||||
|
||||
{OPDReg(0xDA, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDA, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDA, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xDA, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDA, 0xC8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDA, 0xD0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDA, 0xD8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
// E0 = Invalid
|
||||
// E8 = Invalid
|
||||
{OPD(0xDA, 0xE9), 1, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>},
|
||||
// EA = Invalid
|
||||
// F0 = Invalid
|
||||
// F8 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 0) | 0x00, 8, &OpDispatchBuilder::FILDF64},
|
||||
|
||||
{OPDReg(0xDB, 1) | 0x00, 8, &OpDispatchBuilder::FISTF64<true>},
|
||||
|
||||
{OPDReg(0xDB, 2) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
{OPDReg(0xDB, 3) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
// 4 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 5) | 0x00, 8, &OpDispatchBuilder::FLDF64<80>},
|
||||
|
||||
// 6 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::FSTF64<80>},
|
||||
|
||||
|
||||
{OPD(0xDB, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDB, 0xC8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDB, 0xD0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDB, 0xD8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
// E0 = Invalid
|
||||
{OPD(0xDB, 0xE2), 1, &OpDispatchBuilder::NOPOp}, // FNCLEX
|
||||
{OPD(0xDB, 0xE3), 1, &OpDispatchBuilder::FNINITF64},
|
||||
// E4 = Invalid
|
||||
{OPD(0xDB, 0xE8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDB, 0xF0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
|
||||
// F8 = Invalid
|
||||
|
||||
{OPDReg(0xDC, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDC, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDC, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xDC, 0xC0), 8, &OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xC8), 8, &OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xE0), 8, &OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xE8), 8, &OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xF0), 8, &OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xF8), 8, &OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
|
||||
{OPDReg(0xDD, 0) | 0x00, 8, &OpDispatchBuilder::FLDF64<64>},
|
||||
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::FISTF64<true>},
|
||||
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::FSTF64<64>},
|
||||
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::FSTF64<64>},
|
||||
|
||||
{OPDReg(0xDD, 4) | 0x00, 8, &OpDispatchBuilder::X87FRSTORF64},
|
||||
|
||||
// 5 = Invalid
|
||||
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVEF64},
|
||||
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
|
||||
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
|
||||
{OPD(0xDD, 0xD0), 8, &OpDispatchBuilder::FST}, //register-register from regular X87
|
||||
{OPD(0xDD, 0xD8), 8, &OpDispatchBuilder::FST}, //^
|
||||
|
||||
{OPD(0xDD, 0xE0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
{OPD(0xDD, 0xE8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDE, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDE, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDE, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xDE, 0xC0), 8, &OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xC8), 8, &OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xD9), 1, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>},
|
||||
{OPD(0xDE, 0xE0), 8, &OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xE8), 8, &OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xF0), 8, &OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xF8), 8, &OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
|
||||
{OPDReg(0xDF, 0) | 0x00, 8, &OpDispatchBuilder::FILDF64},
|
||||
|
||||
{OPDReg(0xDF, 1) | 0x00, 8, &OpDispatchBuilder::FISTF64<true>},
|
||||
|
||||
{OPDReg(0xDF, 2) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
{OPDReg(0xDF, 3) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
{OPDReg(0xDF, 4) | 0x00, 8, &OpDispatchBuilder::FBLDF64},
|
||||
|
||||
{OPDReg(0xDF, 5) | 0x00, 8, &OpDispatchBuilder::FILDF64},
|
||||
|
||||
{OPDReg(0xDF, 6) | 0x00, 8, &OpDispatchBuilder::FBSTPF64},
|
||||
|
||||
{OPDReg(0xDF, 7) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
// XXX: This should also set the x87 tag bits to empty
|
||||
// We don't support this currently, so just pop the stack
|
||||
{OPD(0xDF, 0xC0), 8, &OpDispatchBuilder::X87ModifySTP<true>},
|
||||
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPD(0xDF, 0xE8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDF, 0xF0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> X87OpTable[] = {
|
||||
{OPDReg(0xD8, 0) | 0x00, 8, &OpDispatchBuilder::FADD<32, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
@@ -6110,7 +6520,6 @@ constexpr uint16_t PF_F2 = 3;
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F38Table[] = {
|
||||
@@ -6156,7 +6565,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<4, 8, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<4, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VCMPEQ, 8>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<4>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, false>},
|
||||
@@ -6176,24 +6585,13 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMUL, 4>},
|
||||
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
|
||||
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
|
||||
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
|
||||
{OPD(PF_38_66, 0xDF), 1, &OpDispatchBuilder::AESDecLastOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
{OPD(PF_38_66, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
|
||||
{OPD(PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
{OPD(PF_38_F3, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
@@ -6225,8 +6623,9 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<4>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
@@ -6308,6 +6707,8 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(2, 0b10, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
{OPD(2, 0b11, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
|
||||
{OPD(3, 0b01, 0x44), 1, &OpDispatchBuilder::VPCLMULQDQOp},
|
||||
|
||||
{OPD(3, 0b11, 0xF0), 1, &OpDispatchBuilder::RORX},
|
||||
};
|
||||
#undef OPD
|
||||
@@ -6372,10 +6773,18 @@ constexpr uint16_t PF_F2 = 3;
|
||||
InstallToTable(FEXCore::X86Tables::RepNEModOps, RepNEModOpTable);
|
||||
InstallToTable(FEXCore::X86Tables::OpSizeModOps, OpSizeModOpTable);
|
||||
InstallToTable(FEXCore::X86Tables::SecondInstGroupOps, SecondaryExtensionOpTable);
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
InstallToTable(FEXCore::X86Tables::SecondInstGroupOps, SecondaryExtensionOpTable_64);
|
||||
}
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::SecondModRMTableOps, SecondaryModRMExtensionOpTable);
|
||||
|
||||
InstallToX87Table(FEXCore::X86Tables::X87Ops, X87OpTable);
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
if(ReducedPrecision) {
|
||||
InstallToX87Table(FEXCore::X86Tables::X87Ops, X87F64OpTable);
|
||||
} else {
|
||||
InstallToX87Table(FEXCore::X86Tables::X87Ops, X87OpTable);
|
||||
}
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38Table);
|
||||
InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3ATable);
|
||||
|
||||
Loaded 100 of 635 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user