mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 18:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ae34b1e521 | ||
|
|
c17da25617 | ||
|
|
cc8ef16240 | ||
|
|
30fac81c41 | ||
|
|
bbcca80b60 | ||
|
|
56ae3716bb | ||
|
|
2fba3a4a9a | ||
|
|
8ba0312d40 | ||
|
|
53623ffa72 | ||
|
|
d6199687f4 | ||
|
|
7cb413dde6 | ||
|
|
c5a7fc1e2a | ||
|
|
0869b0aa29 | ||
|
|
4e81f847f2 | ||
|
|
8fa27d2d9f | ||
|
|
e5a8a29efe | ||
|
|
f967f53176 | ||
|
|
31fefaae0d | ||
|
|
7f9edbf39e | ||
|
|
760b9c8e7f | ||
|
|
cc7fb008fc | ||
|
|
98dbfbe654 | ||
|
|
6444726614 | ||
|
|
0a562fbb10 | ||
|
|
7d8950de40 | ||
|
|
3fdde0c90b | ||
|
|
e776f4cd4e | ||
|
|
e7d7dd13d7 | ||
|
|
d5c83a2e45 | ||
|
|
37ccb13917 | ||
|
|
8439cf410f | ||
|
|
614d9448f7 | ||
|
|
70efdbba5d | ||
|
|
12fee91ebb | ||
|
|
bcb7e20619 | ||
|
|
afc5e8a140 | ||
|
|
4be6626c89 | ||
|
|
5f2b6d629b | ||
|
|
1b8f5f08f0 | ||
|
|
79674c697a | ||
|
|
416d8c1df6 | ||
|
|
d03b6a9382 | ||
|
|
df22e0c796 | ||
|
|
79a3bd75cc | ||
|
|
e6acdcc583 | ||
|
|
cd78984228 | ||
|
|
2f007c5f0a | ||
|
|
ae64a1e30c | ||
|
|
097184c3e0 | ||
|
|
84a95adae1 | ||
|
|
123b6728e4 | ||
|
|
69b4fc98ed | ||
|
|
35cf7703b1 | ||
|
|
8c1137543b | ||
|
|
b8e66e56a0 | ||
|
|
fbb008e510 | ||
|
|
04678f8404 | ||
|
|
1b5aaf1fb8 | ||
|
|
d8e4873f43 | ||
|
|
998a3d8353 | ||
|
|
336dedbed5 | ||
|
|
955595be8a | ||
|
|
576bd4f69a | ||
|
|
cd6915917b | ||
|
|
0adbe31112 | ||
|
|
d5138f509c | ||
|
|
1fe6fc3feb | ||
|
|
c03a7fd482 | ||
|
|
31c47f05f4 | ||
|
|
edad24479b | ||
|
|
bba58732e0 | ||
|
|
8d1cc6cf51 | ||
|
|
f03d0be5ef | ||
|
|
a2f4f494a9 | ||
|
|
9de25c200c | ||
|
|
fe1f00aadb | ||
|
|
a847fac4ca | ||
|
|
0496506fb7 | ||
|
|
e544591c9c | ||
|
|
e5237d1149 | ||
|
|
b9c848c5e9 | ||
|
|
b64e61a793 | ||
|
|
e80e2bdafe | ||
|
|
ac23bce0ba | ||
|
|
1051cd97cf | ||
|
|
45330fdd5d | ||
|
|
bd296d7a11 | ||
|
|
249e19bf2a | ||
|
|
baf52ed286 | ||
|
|
dd7e1baa78 | ||
|
|
80a209d6dc | ||
|
|
ab228c1fcb | ||
|
|
0dde233b1e | ||
|
|
cf0d92d968 | ||
|
|
ce4b052242 | ||
|
|
125bb3afe1 | ||
|
|
30ee2c18ef | ||
|
|
ff1a5dd4c2 | ||
|
|
b9c9c7d671 | ||
|
|
62d9961bd1 | ||
|
|
4b7ed9599d | ||
|
|
c8951de9dd | ||
|
|
63517c377d | ||
|
|
5005ebdfc9 | ||
|
|
a2615f0e52 | ||
|
|
0317d381fe | ||
|
|
31b5181bca | ||
|
|
7ae59055ff | ||
|
|
3524117c23 | ||
|
|
e8ad1ca0a0 | ||
|
|
3ac1650001 | ||
|
|
4f8ae81562 | ||
|
|
329d624a99 | ||
|
|
de92624c9a | ||
|
|
c6a034da40 | ||
|
|
737e76968a | ||
|
|
7c6d49155c | ||
|
|
4d404ea94d | ||
|
|
dc810c7a1e | ||
|
|
e0ee4e71f8 | ||
|
|
ab7dac90a2 | ||
|
|
3a423fbc41 | ||
|
|
0982ec617d | ||
|
|
b45f27c8d0 | ||
|
|
54f62b6701 | ||
|
|
9664d98ba5 | ||
|
|
0c318834b5 | ||
|
|
4c2836f4b3 | ||
|
|
a112db169c | ||
|
|
0dc33ec893 | ||
|
|
0e93ba532f | ||
|
|
9e524a30d6 | ||
|
|
ecca4e4abc | ||
|
|
51791e9efa | ||
|
|
5eab087cc2 | ||
|
|
cd05cdaa57 | ||
|
|
f5e18ccea6 | ||
|
|
0da6b07770 | ||
|
|
05805356fb | ||
|
|
a72ebfdec0 | ||
|
|
59e5da9c7f | ||
|
|
cfd59db998 | ||
|
|
1fa6bf1cc8 | ||
|
|
a2f89de71e | ||
|
|
36a27de286 | ||
|
|
9b685ba824 | ||
|
|
8e9d5fe6ac | ||
|
|
89aa590615 | ||
|
|
3960e0f2a3 | ||
|
|
c3c52d01d3 | ||
|
|
0caef59e6a | ||
|
|
99e6924ed1 | ||
|
|
431e1629b0 | ||
|
|
ff9f702e52 | ||
|
|
6903159f30 | ||
|
|
504b7a03ad | ||
|
|
6933c2aaff | ||
|
|
5fd00e31bb | ||
|
|
a984674dae | ||
|
|
8589119725 | ||
|
|
5e0205378b | ||
|
|
bff2f2e5f9 | ||
|
|
5f9052a675 | ||
|
|
8691b3964f | ||
|
|
4dfe0a0595 | ||
|
|
164299ce25 | ||
|
|
9269ca6ebe | ||
|
|
93fe22e045 | ||
|
|
3849278d5a | ||
|
|
c0f976e200 | ||
|
|
3143749f64 | ||
|
|
1704805e32 | ||
|
|
3aabe07720 | ||
|
|
c5815ab59e | ||
|
|
f6fcfbabd5 | ||
|
|
a0835c95e4 | ||
|
|
ce0c24eee9 | ||
|
|
c9f0ecb1e8 | ||
|
|
50febb3459 | ||
|
|
601b0f96b1 | ||
|
|
018661609a | ||
|
|
98d935d972 | ||
|
|
b07660c4fa | ||
|
|
8037231376 | ||
|
|
6e422a8691 | ||
|
|
3d347ed565 | ||
|
|
13446d8b6b | ||
|
|
08c34bf573 | ||
|
|
3a64ea1650 | ||
|
|
db3439b61b | ||
|
|
7814be7467 | ||
|
|
8a7e0432f1 | ||
|
|
ff1d51c7bd | ||
|
|
f2602e3136 | ||
|
|
313ce0fed8 | ||
|
|
ba0887defa | ||
|
|
063841f491 | ||
|
|
4aa5c78150 | ||
|
|
c1688fa392 | ||
|
|
0dd03e9cb6 | ||
|
|
8d60d70553 | ||
|
|
c17340547c | ||
|
|
34aa6bf6c7 | ||
|
|
82a6ce63b4 | ||
|
|
3ac6ba0fe2 | ||
|
|
8b0fb4fa19 | ||
|
|
9d437b8863 | ||
|
|
f72469c80a | ||
|
|
eec7972808 | ||
|
|
fc9343e7a4 | ||
|
|
be8e353d12 | ||
|
|
53217acdd0 | ||
|
|
c1e0ea5e0a | ||
|
|
467afc7248 | ||
|
|
62753071f4 | ||
|
|
de32de9edd | ||
|
|
aa7d2c3f7e | ||
|
|
2a1dd68583 | ||
|
|
80909eaa84 | ||
|
|
c6a557abe2 | ||
|
|
8d7b8729f1 | ||
|
|
3f9a1c3751 | ||
|
|
0dfe617d96 | ||
|
|
235ad441b4 | ||
|
|
a9d00b3f8d | ||
|
|
fb41ba172d | ||
|
|
aec5b21d2a | ||
|
|
124097d563 | ||
|
|
e137c2edff | ||
|
|
2b35dd829c | ||
|
|
686895802a | ||
|
|
72d0228cd7 | ||
|
|
298e6cad0c | ||
|
|
ad34c228e3 | ||
|
|
66f74a431f | ||
|
|
57bae90c69 | ||
|
|
83c2e52ba5 | ||
|
|
4f8525da4d | ||
|
|
3b8491b558 | ||
|
|
f69c53d294 | ||
|
|
6b226dd6af | ||
|
|
227462e4d8 | ||
|
|
5f00ba5cae | ||
|
|
0b8a6d9599 | ||
|
|
13bf04a81e | ||
|
|
7eb8409f71 | ||
|
|
e821072cb9 | ||
|
|
9c01dd9d9f | ||
|
|
6a43db8c8f | ||
|
|
3ffc301dd0 | ||
|
|
46fcbe2fc0 | ||
|
|
b7806e47e9 | ||
|
|
a1ed54adc7 | ||
|
|
250a4ea4e3 | ||
|
|
b3e090c8ff | ||
|
|
982518d3a4 | ||
|
|
9ad1d5548d | ||
|
|
0de2558ef7 | ||
|
|
3e0e601616 | ||
|
|
8b202b0ebf | ||
|
|
6d2f98a379 | ||
|
|
84379b5fdf | ||
|
|
d005fdcd03 | ||
|
|
a97fb2f34f | ||
|
|
d48981b6b0 | ||
|
|
af32228e38 | ||
|
|
8b35275ec1 | ||
|
|
302a6c96ff | ||
|
|
f4a4b2c14b | ||
|
|
88b94bef54 | ||
|
|
751b66d45d | ||
|
|
ae6a57e667 | ||
|
|
9110546d34 | ||
|
|
1f1d0706dd | ||
|
|
a82d41ee22 | ||
|
|
e8e70828d1 | ||
|
|
b7a58fdb8a | ||
|
|
dc9fde8fae | ||
|
|
c0cf4f6a61 | ||
|
|
21fc6bedcb | ||
|
|
ad6fd5ab72 | ||
|
|
0f696c6092 | ||
|
|
9af3bc1864 | ||
|
|
b020e593a5 | ||
|
|
4771a340f5 | ||
|
|
4449b60459 | ||
|
|
4139332ad9 | ||
|
|
30a28ff1ad | ||
|
|
498d0fc145 | ||
|
|
4a547fe95f | ||
|
|
ca906589d4 | ||
|
|
8bafae2262 | ||
|
|
4ef82c82b5 | ||
|
|
e03d253310 | ||
|
|
48c45da2de | ||
|
|
04a1ac967c | ||
|
|
aa17f64593 | ||
|
|
ae5cfcc249 | ||
|
|
8f578b57f2 | ||
|
|
3871646611 | ||
|
|
48a574dfd3 | ||
|
|
ec49100c63 | ||
|
|
c04d2409da | ||
|
|
b93b713179 | ||
|
|
fe2f54fc3d | ||
|
|
599f8f99ed | ||
|
|
b1b338ed5a | ||
|
|
e4d659a619 | ||
|
|
9ce94266e1 | ||
|
|
5a19425b28 | ||
|
|
1494aac861 | ||
|
|
8a21ecabee | ||
|
|
005b2bc3db | ||
|
|
f27830bf41 | ||
|
|
5fe6afc270 | ||
|
|
4f9bc70562 | ||
|
|
835155021d | ||
|
|
e2c5c2292d | ||
|
|
072690a191 | ||
|
|
a2c9d5a398 | ||
|
|
5944342604 | ||
|
|
4dbcd34548 | ||
|
|
f02c73a2c2 | ||
|
|
158ba1ae3b | ||
|
|
a85a77e088 | ||
|
|
542ab04671 | ||
|
|
28ee2ca5a2 | ||
|
|
3913dd6c8b | ||
|
|
1ecf147e3e | ||
|
|
d8fa53a445 | ||
|
|
1c1ad876ca | ||
|
|
5cd0f6e7e7 | ||
|
|
2627ed3193 | ||
|
|
e3bb8b43ab | ||
|
|
8ba6a3bda9 | ||
|
|
d288249609 | ||
|
|
33d7c6be9e | ||
|
|
c13acc5b60 | ||
|
|
fc9be28d51 | ||
|
|
944f400d19 | ||
|
|
5aee30a7b5 | ||
|
|
58ae49372c | ||
|
|
917e69b021 | ||
|
|
9242e59841 | ||
|
|
05d15fa052 | ||
|
|
0a62a4c571 | ||
|
|
525266e482 | ||
|
|
503881f5df | ||
|
|
04fd4f2bb4 | ||
|
|
8be1e5e260 | ||
|
|
e84e084d56 | ||
|
|
12ad05e089 | ||
|
|
472ce7c1e4 | ||
|
|
19fbc10c16 | ||
|
|
39f3406fba | ||
|
|
19b0a9cd7d | ||
|
|
eac579f714 | ||
|
|
b0a31f7105 | ||
|
|
29163cfd07 | ||
|
|
f05b24636f | ||
|
|
39466a3a8f | ||
|
|
bccf18f46c | ||
|
|
1109126e35 | ||
|
|
98dd40c3a8 | ||
|
|
b1cfb104ff | ||
|
|
c339cfe184 | ||
|
|
ca8e6a83e1 | ||
|
|
760217ab29 | ||
|
|
ee9361bf41 | ||
|
|
e62bc24b3f | ||
|
|
18074307f6 | ||
|
|
63b70ff3d4 | ||
|
|
dacdfd5c02 | ||
|
|
d5c039cdc7 | ||
|
|
e2e6f2a92b | ||
|
|
d7d8244593 | ||
|
|
b05e5ce14e | ||
|
|
08730c817a | ||
|
|
34e086eec2 | ||
|
|
83c8a9b675 | ||
|
|
a3db629fd6 | ||
|
|
64b3cef126 | ||
|
|
7100a2eaee | ||
|
|
ffcde1823b | ||
|
|
4c73c715ad | ||
|
|
26c3ebd6a7 | ||
|
|
790bd9747f | ||
|
|
83aa8731d1 | ||
|
|
bbd9eb5b9a | ||
|
|
cc90fa5773 | ||
|
|
30ebc6c938 | ||
|
|
c0a8984799 | ||
|
|
a6e34d301e | ||
|
|
82c88168cc | ||
|
|
d169c3ed4a | ||
|
|
563d11a702 | ||
|
|
31e2ba4696 | ||
|
|
2c3baaad1c | ||
|
|
01bd43aa2b | ||
|
|
f4d095ce0f | ||
|
|
8c5bd9a058 | ||
|
|
335ce1d056 | ||
|
|
0afb3caaed | ||
|
|
8521aaf65c | ||
|
|
0a451cfe0d | ||
|
|
cb0935c96b | ||
|
|
40ec910108 | ||
|
|
2d3c6efae2 | ||
|
|
c027acecf8 | ||
|
|
bc2840e4a7 | ||
|
|
869b472d91 | ||
|
|
81dfc21700 | ||
|
|
a3d8fe2362 | ||
|
|
d2a57f6231 | ||
|
|
9e9ceb3894 | ||
|
|
bd70088877 | ||
|
|
63c9d99b33 | ||
|
|
a2469f48e7 | ||
|
|
62f0f421a9 | ||
|
|
0b24758f41 | ||
|
|
147897e13d | ||
|
|
ade3a5275d | ||
|
|
51c5f945b4 | ||
|
|
a4e2e36243 | ||
|
|
2c5dc13201 | ||
|
|
43234939ca | ||
|
|
87b3d50899 | ||
|
|
da8dbf1777 | ||
|
|
34e1fcccf8 | ||
|
|
c99d1e48bd | ||
|
|
6a428043f6 | ||
|
|
aafe7ff10f | ||
|
|
952e157770 | ||
|
|
cae4f2f873 | ||
|
|
0fc6d6b6b5 | ||
|
|
7227ee9b2e | ||
|
|
c6153d6a52 | ||
|
|
d82d2944a9 | ||
|
|
58ad400519 | ||
|
|
d23c76d0a9 | ||
|
|
6a5b9e2a93 | ||
|
|
f33a93b0a1 | ||
|
|
015200f511 | ||
|
|
92f48819b6 | ||
|
|
0c6483cad5 | ||
|
|
ee02b1ca51 | ||
|
|
33845a3112 | ||
|
|
096ed29b5e | ||
|
|
3bbff8a948 | ||
|
|
726918b82c | ||
|
|
0f59a18223 | ||
|
|
3402cde334 | ||
|
|
8f53c6bb96 | ||
|
|
0d6e4631a3 | ||
|
|
8dd9a5bd38 | ||
|
|
903cf84874 | ||
|
|
e997da48c7 | ||
|
|
5ff89fd171 | ||
|
|
fad4254c0e | ||
|
|
2fb3c4f11c | ||
|
|
ce5297b75f | ||
|
|
b2b0c277f6 | ||
|
|
46919979ce | ||
|
|
8b716c6a22 | ||
|
|
75090f8f6c | ||
|
|
c14c0c2e3b | ||
|
|
95efd18b73 | ||
|
|
c9319a768f | ||
|
|
da48020882 | ||
|
|
ce4380e136 | ||
|
|
c633661121 | ||
|
|
4bfd1dde1f | ||
|
|
fe11bd2242 | ||
|
|
814f0c3c93 | ||
|
|
7e904056d3 | ||
|
|
969d8f866c | ||
|
|
a2d0b7d7c4 | ||
|
|
97a8fa77bc | ||
|
|
c1296cc64d | ||
|
|
1dee54a9d8 | ||
|
|
17e5d73e64 | ||
|
|
d523b7a6c7 | ||
|
|
24ad208778 | ||
|
|
fa87c73b9e | ||
|
|
0ed96544e1 | ||
|
|
d28ccc59ac | ||
|
|
eaa75c1ed2 | ||
|
|
4f4263263b | ||
|
|
ae00654694 | ||
|
|
a7156276e9 | ||
|
|
29859d2491 | ||
|
|
5460a24ea9 | ||
|
|
a284adcd19 | ||
|
|
73d43c1d55 | ||
|
|
256df76674 | ||
|
|
c8dc663b0b | ||
|
|
ba78dff1f8 | ||
|
|
1e597bfbed | ||
|
|
b78af2fdaf | ||
|
|
5379f0a9c7 | ||
|
|
f1f523e525 | ||
|
|
c79d79e08b | ||
|
|
ee2d417d21 | ||
|
|
cebdde599a | ||
|
|
c3ac72a01e | ||
|
|
13f3c6e75a | ||
|
|
b3cd4edb3b | ||
|
|
afe10c1666 | ||
|
|
d9d30916ba | ||
|
|
3f6c1c0e68 | ||
|
|
5b2cc77109 | ||
|
|
b824023ec6 | ||
|
|
6ce1be0880 | ||
|
|
c5dacab2ee | ||
|
|
4d24b85d57 | ||
|
|
e967b447e6 | ||
|
|
9bc631a427 | ||
|
|
c480ef137d | ||
|
|
15629e790e | ||
|
|
3f08d8b691 | ||
|
|
9d9d171aad | ||
|
|
65218c8285 | ||
|
|
27f2e0b06d | ||
|
|
ae75983b54 | ||
|
|
f8ba373e18 | ||
|
|
2feae06209 | ||
|
|
2e0534924a | ||
|
|
ad1fd7f54b | ||
|
|
9aaace51e1 | ||
|
|
58841142ee | ||
|
|
a6a816fb38 | ||
|
|
70988ccfee | ||
|
|
a8d9caf0c0 | ||
|
|
bc22186093 | ||
|
|
317416b2e0 | ||
|
|
b5ae9e4c97 | ||
|
|
099737ca05 | ||
|
|
e90164b519 | ||
|
|
933c1af7e8 | ||
|
|
b9d878b1f4 | ||
|
|
45a9a83c79 | ||
|
|
560cfc757c | ||
|
|
e92f51e415 | ||
|
|
278ca52d97 | ||
|
|
912dbfe5bd | ||
|
|
687f46fc71 | ||
|
|
6e9e5b3bd6 | ||
|
|
4fbc266b18 | ||
|
|
d1ac406895 | ||
|
|
ce0f5db6f7 | ||
|
|
efb42c1ad1 | ||
|
|
d8109880f4 | ||
|
|
09be28a443 | ||
|
|
fb0bb8dd2c | ||
|
|
8e36f5331f | ||
|
|
a365a70275 | ||
|
|
b5a4e5920d | ||
|
|
94d2ed85a7 | ||
|
|
90f338d7db | ||
|
|
df78f5d50e | ||
|
|
b2b4c2bdcf | ||
|
|
a888da436b | ||
|
|
3cb8ae9a9c | ||
|
|
db3854e391 | ||
|
|
05b4b095fe | ||
|
|
93926641d9 | ||
|
|
89d6752d3d | ||
|
|
e4f95fec79 | ||
|
|
8a7f39559c | ||
|
|
753d0ede6c | ||
|
|
da2e44d024 | ||
|
|
7b379fc3cf | ||
|
|
ec38d58b37 | ||
|
|
f6a74a710d | ||
|
|
3fd136b0da | ||
|
|
72e82d0304 | ||
|
|
2f7dcb8d93 | ||
|
|
128a24d699 | ||
|
|
cd94a8f0ac | ||
|
|
df5e0e5df9 | ||
|
|
458bbf4ef7 | ||
|
|
82319c9deb | ||
|
|
3bc4df7295 | ||
|
|
42a6320935 | ||
|
|
843fe378db | ||
|
|
3679673d5b | ||
|
|
3a269f04d2 | ||
|
|
0021723b50 | ||
|
|
1c4b0272e8 | ||
|
|
9640216124 | ||
|
|
0f8f2bf2b4 | ||
|
|
ef899b7b1a | ||
|
|
b85abf725d | ||
|
|
3f89e46d66 | ||
|
|
d02712ddc6 | ||
|
|
91f48c63ff | ||
|
|
cd16769e57 | ||
|
|
0a778e802f | ||
|
|
d4d5f4d1dd | ||
|
|
b1033ed7c6 | ||
|
|
253333a4cf | ||
|
|
50595ac3a9 | ||
|
|
37f1e55ed5 | ||
|
|
8ad14728f6 | ||
|
|
6b3cd3d31d | ||
|
|
fba698cb74 | ||
|
|
0946b123bb | ||
|
|
b43937a7a1 | ||
|
|
4564eba20d | ||
|
|
5cc0c0a3da | ||
|
|
f8e7c75f86 | ||
|
|
042cd354dc | ||
|
|
977bda97b2 | ||
|
|
4cf48ca9bb | ||
|
|
a247df50ea | ||
|
|
1f1c214944 | ||
|
|
d16db4ebde | ||
|
|
ae1c563082 | ||
|
|
ebd0edbab7 | ||
|
|
e87e9d269a | ||
|
|
187c64182b | ||
|
|
60c7ea6e5f | ||
|
|
6f1b4b0eee | ||
|
|
fb69300397 | ||
|
|
5677924525 | ||
|
|
8422fc632d | ||
|
|
d33cd744fb | ||
|
|
3b0fb27ae9 | ||
|
|
4603e09a04 | ||
|
|
7b0265ffe2 | ||
|
|
ec54560a38 | ||
|
|
6a5abd3672 | ||
|
|
3e6af39c42 | ||
|
|
cf82ffc052 | ||
|
|
b190150281 | ||
|
|
53ffe5df43 | ||
|
|
d39df8d3ed | ||
|
|
eeb2b928b9 | ||
|
|
6cf24a748f | ||
|
|
3b4fd180de | ||
|
|
7300c7a853 | ||
|
|
376f6db3ac | ||
|
|
ad3a960717 | ||
|
|
0a1ef867ee | ||
|
|
a7fe69deea | ||
|
|
a8c3b6d46f | ||
|
|
60db2655bb | ||
|
|
fd717b6995 | ||
|
|
ba37388fe3 | ||
|
|
68f32d85c9 | ||
|
|
2aa77e85de | ||
|
|
91665fdf0e | ||
|
|
23a1c64bf7 | ||
|
|
c2dcf06632 | ||
|
|
fad91bb818 | ||
|
|
0e769ece26 | ||
|
|
9a780b40a2 | ||
|
|
898873e9e3 | ||
|
|
52292e5f7e | ||
|
|
5de6c866b7 | ||
|
|
ec0cd3aec4 | ||
|
|
f5f9512d9a | ||
|
|
fb27cb4356 | ||
|
|
94664580c8 | ||
|
|
99a93fa9ea | ||
|
|
4cb6918506 | ||
|
|
9cc743bf84 | ||
|
|
a408749eef | ||
|
|
d8a3687ac3 | ||
|
|
a32c7f6ce2 | ||
|
|
540feb857b | ||
|
|
a3902a0d2d | ||
|
|
5fbd01536f | ||
|
|
bebcab0277 | ||
|
|
d0f17d400e | ||
|
|
11a07eb3f2 | ||
|
|
cb13e1bdb9 | ||
|
|
3e8c6d0be0 | ||
|
|
847b6e5026 | ||
|
|
7f6e9d3dae | ||
|
|
e244e8f66b | ||
|
|
0adaa85d97 | ||
|
|
f1d34c7407 | ||
|
|
7a9492ceca | ||
|
|
39192cd092 | ||
|
|
73abf9b9dd | ||
|
|
0387c24e21 | ||
|
|
9d14bc7846 | ||
|
|
ac32ecadbe | ||
|
|
08938ecb11 | ||
|
|
4a3cbf1b1f | ||
|
|
0a0bc21c88 | ||
|
|
a7ad7f4456 | ||
|
|
46ac05e3cf | ||
|
|
9911fe68d4 | ||
|
|
57a56545b2 | ||
|
|
b4e0565907 | ||
|
|
5137af5bae | ||
|
|
385ed2c2ec | ||
|
|
3418cc8054 | ||
|
|
1f835ea1f4 | ||
|
|
0a40753624 | ||
|
|
609538e758 | ||
|
|
09010e1292 | ||
|
|
bddf3871ba | ||
|
|
1ca9e56502 | ||
|
|
fe58a9ae2c | ||
|
|
bc7c4dfa74 | ||
|
|
7af7b1309d | ||
|
|
dd4630750f | ||
|
|
302da029eb | ||
|
|
3cc59bc68e | ||
|
|
30c27851ef | ||
|
|
9b29ff61e9 | ||
|
|
91f780223b | ||
|
|
2933a00b12 | ||
|
|
a0efd2b01f | ||
|
|
3e9dbda146 | ||
|
|
6501715a3c | ||
|
|
2af23d9bec | ||
|
|
3ba2d6cfb4 | ||
|
|
abd266441c | ||
|
|
b1b6078518 | ||
|
|
a9a89bea9f | ||
|
|
9b1b2e6496 | ||
|
|
638da92f45 | ||
|
|
0f3a169cc4 | ||
|
|
c7dd176799 | ||
|
|
23daf4cb72 | ||
|
|
bb7fa84fb8 | ||
|
|
3325ba52b9 | ||
|
|
7f47fe6d73 | ||
|
|
ee165379c5 | ||
|
|
fa554d3096 | ||
|
|
d60710d3f1 | ||
|
|
75988b2ae5 | ||
|
|
30803c66f7 | ||
|
|
a9d838fa27 | ||
|
|
2b8f60c108 | ||
|
|
5ec6ee5b69 | ||
|
|
0db7205f61 | ||
|
|
6027494d69 | ||
|
|
252dcfe26f | ||
|
|
944c93c10c | ||
|
|
a5fb7e7313 | ||
|
|
364b3380fc | ||
|
|
a0edab8040 | ||
|
|
b65194f433 | ||
|
|
0d1c9cd7df | ||
|
|
99dcda7f8c | ||
|
|
2aa4d33de5 | ||
|
|
1ef78d0a00 | ||
|
|
64d1840f1e | ||
|
|
34530236f9 | ||
|
|
f5a9da082f | ||
|
|
9b49e8cb59 | ||
|
|
1b6d20b731 | ||
|
|
e9c7c76174 | ||
|
|
cbce06d012 | ||
|
|
d95326b23b | ||
|
|
43fada7555 | ||
|
|
afa7172cb1 | ||
|
|
b88d8a7cc4 | ||
|
|
6add09b78b | ||
|
|
4bb3a54ccf | ||
|
|
76f86e51a6 | ||
|
|
2a1b27df58 | ||
|
|
68426735a5 | ||
|
|
ee6b558000 | ||
|
|
3c7872c6d5 | ||
|
|
4aed6fc56c | ||
|
|
5d555a10a2 | ||
|
|
5854d4ad1c | ||
|
|
bd6999eb87 | ||
|
|
3ebb2eaf0d | ||
|
|
defd3be30c | ||
|
|
a8e5a0a68e | ||
|
|
f27c43ff8a | ||
|
|
8840fc818c | ||
|
|
ffcaf294e3 | ||
|
|
6b7a84bef2 | ||
|
|
287c65dc64 | ||
|
|
62397bb19c | ||
|
|
0216bcf27f | ||
|
|
b981fcfe42 | ||
|
|
cb491a8acb | ||
|
|
1b99495b4b | ||
|
|
3c5a2cec90 | ||
|
|
9a64e7f100 | ||
|
|
8e8baec47a | ||
|
|
59ca60e39f | ||
|
|
8c956e6ce1 | ||
|
|
832d013c92 | ||
|
|
1c24206117 | ||
|
|
f1979c15a2 | ||
|
|
556a1dab24 | ||
|
|
caffad8562 | ||
|
|
5978143141 | ||
|
|
594c70b5e0 | ||
|
|
655e6989ca | ||
|
|
afeb228a89 | ||
|
|
d308a438ea | ||
|
|
260fc8ba52 | ||
|
|
ade0d0f241 | ||
|
|
11a5105547 | ||
|
|
10ad5db686 | ||
|
|
68c441575d | ||
|
|
334a8ef87c | ||
|
|
b7a76af72f | ||
|
|
4c92b562b8 | ||
|
|
9d08451903 | ||
|
|
c252f8bfc5 | ||
|
|
dc7ec6377b | ||
|
|
ce6f4edaaa | ||
|
|
983c35ea3b | ||
|
|
70aaa1117a | ||
|
|
59e9859087 | ||
|
|
d9453ff639 | ||
|
|
70754991d1 | ||
|
|
57ebfceb48 | ||
|
|
9e224d2bb0 | ||
|
|
e43bd04901 | ||
|
|
48762e03a6 | ||
|
|
858924309e | ||
|
|
cab02d1e65 | ||
|
|
2a64f80567 | ||
|
|
174ddea99d | ||
|
|
6fb0e3c0cf | ||
|
|
2b044bbdf4 | ||
|
|
ea76de0fd2 | ||
|
|
9c8642e0dc | ||
|
|
13f35f7b79 | ||
|
|
e2798e370e | ||
|
|
d9149548b5 | ||
|
|
4823933f79 | ||
|
|
ee04067424 | ||
|
|
e4aef26ef5 | ||
|
|
a41dc8eafa | ||
|
|
f41cd8deff | ||
|
|
2a0c3cce30 | ||
|
|
e817f5d98c | ||
|
|
8b8cda9b80 | ||
|
|
8e3893df07 | ||
|
|
51b335914c | ||
|
|
8835d57ae3 | ||
|
|
eba1b65fb0 | ||
|
|
a3a138ef7e | ||
|
|
62d9a494cd | ||
|
|
6744a06a53 | ||
|
|
140e9824b7 | ||
|
|
bad84f61fa | ||
|
|
2296126af3 | ||
|
|
0a8717d9a8 | ||
|
|
4e2220c27f | ||
|
|
c87e11cee9 | ||
|
|
7768f6965a | ||
|
|
023aaaae0c | ||
|
|
a2aa9f3fc1 | ||
|
|
e46ec9a0ce | ||
|
|
784cbdd973 | ||
|
|
6022715a9b | ||
|
|
9eb5ba5ad1 | ||
|
|
82e5977709 | ||
|
|
7c08b67dff | ||
|
|
609587f9ee | ||
|
|
5dda3a1599 | ||
|
|
73aaa4c3a6 | ||
|
|
3ba5371d36 | ||
|
|
bf581decde | ||
|
|
250504502a | ||
|
|
a0e826feaf | ||
|
|
eb17edec05 | ||
|
|
228aed98c7 | ||
|
|
6ba4aec88e | ||
|
|
b8d6b2cd4a | ||
|
|
d4e2f42f90 | ||
|
|
3055c23365 | ||
|
|
cb7feaefbb | ||
|
|
5a5a498ed6 | ||
|
|
285ed8f1e0 | ||
|
|
a68da468a5 | ||
|
|
d59aa6874e | ||
|
|
19fd89d2bf | ||
|
|
e36beb8dbe | ||
|
|
70f447b265 | ||
|
|
a0026c92a8 | ||
|
|
8fc5f66a5b |
No files matched your search
@@ -29,9 +29,24 @@ jobs:
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: git submodule update --init --depth 1
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
@@ -49,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -151,6 +166,28 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
- name: Thunkgen tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
|
||||
|
||||
- name: Thunkgen Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
@@ -10,3 +10,4 @@ out/
|
||||
.vscode/
|
||||
.vs/
|
||||
*.pyc
|
||||
.cache
|
||||
+9
-1
@@ -1,7 +1,7 @@
|
||||
[submodule "External/vixl"]
|
||||
shallow = true
|
||||
path = External/vixl
|
||||
url = https://github.com/Sonicadvance1/vixl.git
|
||||
url = https://github.com/FEX-Emu/vixl.git
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = External/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
@@ -45,3 +45,11 @@
|
||||
[submodule "External/Catch2"]
|
||||
path = External/Catch2
|
||||
url = https://github.com/catchorg/Catch2.git
|
||||
[submodule "External/robin-map"]
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/Tessil/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
+75
-175
@@ -1,31 +1,39 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX)
|
||||
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with LLD" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_STATIC_PIE "Enables static-pie build" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
|
||||
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
|
||||
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
|
||||
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
|
||||
@@ -38,6 +46,17 @@ if (ENABLE_ASSERTIONS)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GDB_SYMBOLS)
|
||||
message(STATUS "GDBSymbols support enabled")
|
||||
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
message(STATUS "Interpreter enabled")
|
||||
add_definitions(-DINTERPRETER_ENABLED=1)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
|
||||
@@ -65,6 +84,7 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
@@ -72,10 +92,12 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
message(STATUS "CCache enabled")
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
if (ENABLE_CCACHE)
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
message(STATUS "CCache enabled")
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_XRAY)
|
||||
@@ -89,9 +111,14 @@ if (ENABLE_COMPILE_TIME_TRACE)
|
||||
endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
if (ENABLE_LLD)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
link_libraries(${LD_OVERRIDE})
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
@@ -105,128 +132,11 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_STATIC_PIE)
|
||||
if (_M_ARM_64 AND ENABLE_LLD)
|
||||
message (FATAL_ERROR "Static linking does not currently work with AArch64+LLD. Use GNU ld for now.")
|
||||
endif()
|
||||
|
||||
file(WRITE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
"int main(int argc, char* argv[])
|
||||
{
|
||||
return 0;
|
||||
}")
|
||||
|
||||
# Compile the test application with our LD_OVERRIDE and static-pie options
|
||||
try_compile(
|
||||
COMPILE_RESULT
|
||||
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp
|
||||
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
COMPILE_DEFINITIONS "-fPIE ${LD_OVERRIDE}"
|
||||
LINK_LIBRARIES "-static-pie ${LD_OVERRIDE}"
|
||||
COPY_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
|
||||
)
|
||||
|
||||
if (${COMPILE_RESULT})
|
||||
# Read the symbols from the elf
|
||||
execute_process(COMMAND
|
||||
readelf -s ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
|
||||
OUTPUT_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
|
||||
OUTPUT_VARIABLE PLT_SYMBOLS)
|
||||
|
||||
# Pull out the __rela_iplt_{start,end} symbols if they exist
|
||||
execute_process(COMMAND
|
||||
"grep" "__rela_iplt" ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
|
||||
OUTPUT_VARIABLE PLT_SYMBOLS)
|
||||
|
||||
set (SYMBOLS_FINE TRUE)
|
||||
set (HAS_IPLT -1)
|
||||
# Check if we have any symbols in our grep output
|
||||
# The symbols must either not exist at all OR the symbols are zero
|
||||
if (PLT_SYMBOLS)
|
||||
string(FIND ${PLT_SYMBOLS} "__rela_iplt_start" HAS_IPLT)
|
||||
endif()
|
||||
|
||||
if (NOT HAS_IPLT EQUAL -1)
|
||||
# We have some symbols from readelf. Let's parse the results to check if they are zero
|
||||
# Format: '35: 0000000000000000 0 NOTYPE LOCAL HIDDEN UND __rela_iplt_start'
|
||||
string(REPLACE "\n" ";" SYMBOL_LIST ${PLT_SYMBOLS})
|
||||
foreach (SYMBOL ${SYMBOL_LIST})
|
||||
# strip any leading and trailing whitespace
|
||||
string (STRIP ${SYMBOL} SYMBOL)
|
||||
# Convert string to a list
|
||||
string(REPLACE " " ";" SYMBOL_VALUES ${SYMBOL}})
|
||||
# Pull out the address argument
|
||||
list(GET SYMBOL_VALUES 1 OFFSET)
|
||||
|
||||
# Check against integer zero
|
||||
if (NOT ${OFFSET} EQUAL 0)
|
||||
# Symbol wasn't zero, this now fails
|
||||
set (SYMBOLS_FINE FALSE)
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if (SYMBOLS_FINE)
|
||||
# We can now exnable static-pie
|
||||
set (STATIC_PIE_OPTIONS "-static-pie")
|
||||
# Pthreads has an issue with exposing symbols
|
||||
# We need to make some concessions to the pthread gods
|
||||
if (ENABLE_LLD)
|
||||
set (PTHREAD_LIB
|
||||
-Wl,--undefined-glob=pthread_*
|
||||
-Wl,--undefined=__cxa_finalize
|
||||
-Wl,--undefined=_pthread_cleanup_push_defer
|
||||
-Wl,--undefined=_pthread_cleanup_pop_restore
|
||||
-Wl,--undefined=__pthread_cleanup_upto
|
||||
pthread)
|
||||
else()
|
||||
set (PTHREAD_LIB
|
||||
-Wl,--undefined=pthread_join
|
||||
-Wl,--undefined=pthread_attr_getdetachstate
|
||||
-Wl,--undefined=pthread_sigmask
|
||||
-Wl,--undefined=pthread_mutex_lock
|
||||
-Wl,--undefined=pthread_cond_init
|
||||
-Wl,--undefined=pthread_attr_init
|
||||
-Wl,--undefined=pthread_mutex_unlock
|
||||
-Wl,--undefined=pthread_mutexattr_destroy
|
||||
-Wl,--undefined=pthread_detach
|
||||
-Wl,--undefined=pthread_mutex_init
|
||||
-Wl,--undefined=pthread_getattr_np
|
||||
-Wl,--undefined=pthread_cond_timedwait
|
||||
-Wl,--undefined=pthread_attr_destroy
|
||||
-Wl,--undefined=pthread_mutexattr_settype
|
||||
-Wl,--undefined=pthread_rwlock_unlock
|
||||
-Wl,--undefined=pthread_rwlock_wrlock
|
||||
-Wl,--undefined=pthread_setspecific
|
||||
-Wl,--undefined=pthread_create
|
||||
-Wl,--undefined=pthread_cond_clockwait
|
||||
-Wl,--undefined=pthread_key_create
|
||||
-Wl,--undefined=pthread_rwlock_rdlock
|
||||
-Wl,--undefined=pthread_setname_np
|
||||
-Wl,--undefined=pthread_cond_signal
|
||||
-Wl,--undefined=pthread_mutexattr_init
|
||||
-Wl,--undefined=pthread_attr_setstack
|
||||
-Wl,--undefined=pthread_self
|
||||
-Wl,--undefined=pthread_getaffinity_np
|
||||
-Wl,--undefined=pthread_cond_wait
|
||||
-Wl,--undefined=pthread_mutex_trylock
|
||||
-Wl,--undefined=pthread_cond_broadcast
|
||||
-Wl,--undefined=pthread_cond_destroy
|
||||
-Wl,--undefined=pthread_getspecific
|
||||
-Wl,--undefined=pthread_key_delete
|
||||
-Wl,--undefined=pthread_once
|
||||
-Wl,--undefined=__cxa_finalize
|
||||
-Wl,--undefined=_pthread_cleanup_push_defer
|
||||
-Wl,--undefined=_pthread_cleanup_pop_restore
|
||||
-Wl,--undefined=__pthread_cleanup_upto
|
||||
pthread)
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Application has __rela_iplt_{start,end} symbols. Which means static-pie can't be enabled")
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Couldn't compile static-pie test. Static-pie can't be enabled! Is your glibc compiled without static-pie?")
|
||||
endif()
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
@@ -258,6 +168,8 @@ set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fn
|
||||
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(External/vixl/src/)
|
||||
|
||||
@@ -268,13 +180,9 @@ endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
pkg_check_modules(XXHASH libxxhash>=0.8.0 QUIET)
|
||||
|
||||
if (NOT XXHASH_FOUND)
|
||||
message(STATUS "xxHash not found. Using Externals")
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
endif()
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
@@ -335,6 +243,15 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
@@ -457,51 +374,40 @@ add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
# Install the ThunksDB file
|
||||
install(
|
||||
FILES ${CMAKE_CURRENT_SOURCE_DIR}/Data/ThunksDB.json
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
# Thunk targets for both host libraries and IDE integration
|
||||
add_subdirectory(ThunkLibs/HostLibs)
|
||||
|
||||
# Thunk targets for IDE integration of guest code, only
|
||||
add_subdirectory(ThunkLibs/GuestLibs)
|
||||
|
||||
# Thunk targets for guest libraries
|
||||
include(ExternalProject)
|
||||
|
||||
ExternalProject_Add(host-libs
|
||||
PREFIX host-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
|
||||
BINARY_DIR "Host"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: host-libs\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkHostsInstall
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Host
|
||||
)"
|
||||
DEPENDS host-libs
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs
|
||||
PREFIX guest-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DX86_C_COMPILER:STRING=${X86_C_COMPILER}"
|
||||
"-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
@@ -511,7 +417,7 @@ if (BUILD_THUNKS)
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkGuestsInstall
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
@@ -568,15 +474,9 @@ endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
if (ENABLE_STATIC_PIE)
|
||||
set (CPACK_PACKAGE_NAME fex-emu-static)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu")
|
||||
else()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu-static")
|
||||
endif()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.org>")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
|
||||
+1
-1
@@ -55,7 +55,7 @@ further defined and clarified by project maintainers.
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported by contacting the project team at team@fex-emu.org. All
|
||||
reported by contacting the project team at team@fex-emu.com. All
|
||||
complaints will be reviewed and investigated and will result in a response that
|
||||
is deemed necessary and appropriate to the circumstances. The project team is
|
||||
obligated to maintain confidentiality with regard to the reporter of an incident.
|
||||
|
||||
@@ -11,15 +11,15 @@ endforeach()
|
||||
# First generate then install it
|
||||
foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Get the filename only component
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WE)
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
|
||||
|
||||
# Configure it
|
||||
configure_file(
|
||||
${GEN_CONFIG_SRC}
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json)
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
|
||||
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
endforeach()
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"AdditionalArguments": "--no-sandbox"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
+60
-239
@@ -6,18 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -26,322 +18,151 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan-radeon": {
|
||||
"Library": "libvulkan_radeon-guest.so",
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_radeon.so"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
|
||||
]
|
||||
},
|
||||
"Vulkan-lavapipe": {
|
||||
"Library": "libvulkan_lvp-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_lvp.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-freedreno": {
|
||||
"Library": "libvulkan_freedreno-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_freedreno.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-intel": {
|
||||
"Library": "libvulkan_intel-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_intel.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-panfrost": {
|
||||
"Library": "libvulkan_panfrost-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_panfrost.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-nvidia": {
|
||||
"Library": "libvulkan_nvidia-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/lib/x86_64-linux-gnu/libGLX_nvidia.so.0"
|
||||
],
|
||||
"Comment": [
|
||||
"Not currently wired up"
|
||||
]
|
||||
},
|
||||
"Vulkan-virtio": {
|
||||
"Library": "libvulkan_virtio-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_virtio.so"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb_dri2-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb_dri3-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb_xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb_shm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb_sync-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb_randr-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb_present-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb_glx-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
Vendored
+2
-2
@@ -46,14 +46,14 @@ if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
|
||||
Vendored
+2
-12
@@ -18,14 +18,8 @@ This project aims to provide a fast and functional x86-64 emulation library that
|
||||
* Portable library implementation in order to support easy integration in to applications
|
||||
### Target Host Architecture
|
||||
The target host architecture for this library is AArch64. Specifically the ARMv8.1 version or newer.
|
||||
The CPU IR is designed with AArch64 in mind but there is a desire to run the recompiled code on other architectures as well.
|
||||
Multiple architecture support is desired for easier bringup and debugging, performance isn't as much of a priority there (ex. x86-64(guest) translated to x86-64(host))
|
||||
### Not currently goals but will be in the future
|
||||
* 32bit x86 support
|
||||
* This will be a desire in the future, but to lower the amount of work required, decided to push this off for now.
|
||||
* Integration in to WINE
|
||||
* Later generation of x86-64 instruction sets
|
||||
* Including AVX, F16C, XOP, FMA, AVX2, etc
|
||||
The CPU IR is designed with AArch64 in mind but should allow for other architectures as well.
|
||||
x86-64 host support is available for ease of development, but is not a priority.
|
||||
### Not desired
|
||||
* Kernel space emulation
|
||||
* CPL0-2 emulation
|
||||
@@ -33,7 +27,3 @@ Multiple architecture support is desired for easier bringup and debugging, perfo
|
||||
* IRQs
|
||||
* SVM
|
||||
* "Cycle Accurate" emulation
|
||||
### Dependencies
|
||||
* clang-tidy if you want to ensure the code stays tidy
|
||||
* cmake
|
||||
* A C++17 compliant compiler (There are assumptions made about using Clang and LTO)
|
||||
+6
-43
@@ -7,17 +7,12 @@ OpClasses = collections.OrderedDict()
|
||||
def get_ir_classes(ops, defines):
|
||||
global OpClasses
|
||||
|
||||
for op_key, op_vals in ops.items():
|
||||
if not ("Last" in op_vals):
|
||||
OpClass = "#Unknown"
|
||||
for op_class, opslist in ops.items():
|
||||
if not (op_class in OpClasses):
|
||||
OpClasses[op_class] = []
|
||||
|
||||
if ("OpClass" in op_vals):
|
||||
OpClass = op_vals["OpClass"]
|
||||
|
||||
if not (OpClass in OpClasses):
|
||||
OpClasses[OpClass] = []
|
||||
|
||||
OpClasses[OpClass].append([op_key, op_vals])
|
||||
for op, op_val in opslist.items():
|
||||
OpClasses[op_class].append([op, op_val])
|
||||
|
||||
# Sort the dictionary after we are done parsing it
|
||||
OpClasses = collections.OrderedDict(sorted(OpClasses.items()))
|
||||
@@ -38,41 +33,9 @@ def print_ir_ops():
|
||||
op_key = op[0]
|
||||
op_vals = op[1]
|
||||
output_file.write("## %s\n" % (op_key))
|
||||
HasDest = ("HasDest" in op_vals and op_vals["HasDest"] == True)
|
||||
HasSSAArgs = ("SSAArgs" in op_vals and len(op_vals["SSAArgs"]) > 0)
|
||||
HasSSAArgNames = "SSANames" in op_vals
|
||||
HasArgs = "Args" in op_vals
|
||||
SSAArgsCount = 0
|
||||
ArgCount = 0
|
||||
if (HasSSAArgs):
|
||||
SSAArgsCount = int(op_vals["SSAArgs"])
|
||||
if (HasArgs):
|
||||
ArgCount = len(op_vals["Args"])
|
||||
|
||||
TotalArgsCount = SSAArgsCount + (ArgCount / 2)
|
||||
|
||||
output_file.write(">")
|
||||
if (HasDest):
|
||||
output_file.write("%dest = ")
|
||||
|
||||
output_file.write("%s " % op_key)
|
||||
|
||||
ArgComma = (", ", "")
|
||||
if (HasSSAArgs):
|
||||
for i in range(0, SSAArgsCount):
|
||||
FinalArg = (i + 1) == TotalArgsCount
|
||||
if (HasSSAArgNames):
|
||||
output_file.write("%%%s%s" % (op_vals["SSANames"][i], ArgComma[FinalArg]))
|
||||
else:
|
||||
output_file.write("%%ssa%d%s" % (i, ArgComma[FinalArg]))
|
||||
|
||||
if (HasArgs):
|
||||
Args = op_vals["Args"]
|
||||
for i in range(0, ArgCount, 2):
|
||||
FinalArg = ((i / 2) + SSAArgsCount + 1) == TotalArgsCount
|
||||
data_type = Args[i]
|
||||
data_name = Args[i + 1]
|
||||
output_file.write("\<%s %s\>%s" % (data_type, data_name, ArgComma[FinalArg]))
|
||||
output_file.write(op_key)
|
||||
|
||||
output_file.write("\n\n")
|
||||
|
||||
|
||||
+467
-390
File diff suppressed because it is too large.
Load diff
+33
-23
@@ -80,16 +80,20 @@ set (SRCS
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/CompileService.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
Interface/Core/OpcodeDispatcher/Vector.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
@@ -100,19 +104,7 @@ set (SRCS
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/X86Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/ALUOps.cpp
|
||||
Interface/Core/Interpreter/AtomicOps.cpp
|
||||
Interface/Core/Interpreter/BranchOps.cpp
|
||||
Interface/Core/Interpreter/ConversionOps.cpp
|
||||
Interface/Core/Interpreter/EncryptionOps.cpp
|
||||
Interface/Core/Interpreter/F80Ops.cpp
|
||||
Interface/Core/Interpreter/FlagOps.cpp
|
||||
Interface/Core/Interpreter/MemoryOps.cpp
|
||||
Interface/Core/Interpreter/MiscOps.cpp
|
||||
Interface/Core/Interpreter/MoveOps.cpp
|
||||
Interface/Core/Interpreter/VectorOps.cpp
|
||||
Interface/Core/Interpreter/InterpreterFallbacks.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -126,6 +118,7 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
@@ -152,6 +145,23 @@ set (SRCS
|
||||
Utils/Threads.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/ALUOps.cpp
|
||||
Interface/Core/Interpreter/AtomicOps.cpp
|
||||
Interface/Core/Interpreter/BranchOps.cpp
|
||||
Interface/Core/Interpreter/ConversionOps.cpp
|
||||
Interface/Core/Interpreter/EncryptionOps.cpp
|
||||
Interface/Core/Interpreter/F80Ops.cpp
|
||||
Interface/Core/Interpreter/FlagOps.cpp
|
||||
Interface/Core/Interpreter/MemoryOps.cpp
|
||||
Interface/Core/Interpreter/MiscOps.cpp
|
||||
Interface/Core/Interpreter/MoveOps.cpp
|
||||
Interface/Core/Interpreter/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp)
|
||||
@@ -179,7 +189,9 @@ if (ENABLE_JIT_X86_64)
|
||||
Interface/Core/JIT/x86_64/MemoryOps.cpp
|
||||
Interface/Core/JIT/x86_64/MiscOps.cpp
|
||||
Interface/Core/JIT/x86_64/MoveOps.cpp
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp)
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp
|
||||
Interface/Core/JIT/x86_64/x64Relocations.cpp
|
||||
)
|
||||
list(APPEND DEFINES -DJIT_X86_64)
|
||||
endif()
|
||||
|
||||
@@ -196,7 +208,9 @@ if (ENABLE_JIT_ARM64)
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
@@ -214,13 +228,11 @@ set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
add_custom_target(CREATE_IR_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_IR_FOLDER}")
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
)
|
||||
@@ -234,7 +246,6 @@ set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_IR_DOC}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
|
||||
)
|
||||
@@ -255,15 +266,13 @@ set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
|
||||
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
|
||||
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
|
||||
|
||||
add_custom_target(CREATE_CONFIG_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
|
||||
file(MAKE_DIRECTORY "${OUTPUT_CONFIG_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_CONFIG_NAME}"
|
||||
OUTPUT "${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
OUTPUT "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${INPUT_CONFIG_NAME}"
|
||||
DEPENDS CREATE_CONFIG_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
@@ -324,6 +333,7 @@ function(AddDefaultOptionsToTarget Name)
|
||||
|
||||
-Wno-trigraphs
|
||||
-ffunction-sections
|
||||
-fwrapv
|
||||
)
|
||||
|
||||
if (GCC_COLOR)
|
||||
|
||||
+3
-3
@@ -20,12 +20,12 @@ struct BitSet final {
|
||||
ElementType *Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
@@ -64,7 +64,7 @@ struct BitSetView final {
|
||||
ElementType *Memory;
|
||||
|
||||
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0,
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
+17
-6
@@ -7,6 +7,11 @@
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fp.reset(fopen(PerfMap.c_str(), "wb"));
|
||||
@@ -16,14 +21,12 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} JIT_0x{:x}_{:x}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
fmt::print(fp.get(), "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
@@ -31,7 +34,15 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} {}_{:x}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
@@ -39,7 +50,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
fmt::print(fp.get(), "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
|
||||
@@ -47,7 +58,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
fmt::print(fp.get(), "{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+2
@@ -11,8 +11,10 @@ public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
|
||||
|
||||
+269
-9
@@ -57,51 +57,183 @@ struct X80SoftFloat {
|
||||
|
||||
// Ops
|
||||
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
faddp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_add(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fsubp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sub(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fmulp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_mul(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fdivp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_div(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
X80SoftFloat Rem = extF80_rem(lhs, rhs);
|
||||
if (SignBit(Rem)) {
|
||||
Rem = extF80_add(Rem, rhs);
|
||||
}
|
||||
else {
|
||||
Rem.Sign = SignBit(lhs);
|
||||
}
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Rem;
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem1;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
|
||||
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
|
||||
return extF80_roundToInt(lhs, RoundMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
|
||||
@@ -112,61 +244,189 @@ struct X80SoftFloat {
|
||||
|
||||
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
|
||||
X80SoftFloat Int = FRNDINT(rhs);
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fscale; # st0 = st0 * 2^(rdint(st1))
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
|
||||
BIGFLOAT Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
f2xm1; # st0 = 2^st(0) - 1
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st(1)
|
||||
fldt %[lhs]; # st(0)
|
||||
fyl2x; # st(1) * log2l(st(0))
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs];
|
||||
fldt %[rhs];
|
||||
fpatan;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fptan;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsin;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fcos;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsqrt;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sqrt(lhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
operator float() const {
|
||||
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore::StringUtils {
|
||||
// Trim the left side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
// Trim the right side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
// Trim both the left and right of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
|
||||
}
|
||||
}
|
||||
+54
-35
@@ -1,4 +1,5 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Common/Paths.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
@@ -81,7 +82,7 @@ namespace JSON {
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
LogMan::Msg::EFmt("Couldn't get config list");
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -153,15 +154,20 @@ namespace JSON {
|
||||
return ConfigDir;
|
||||
}
|
||||
|
||||
std::string GetConfigFileLocation() {
|
||||
std::string GetConfigFileLocation(bool Global) {
|
||||
std::string ConfigFile{};
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
if (Global) {
|
||||
ConfigFile = GetConfigDirectory(true) + "Config.json";
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
}
|
||||
}
|
||||
return ConfigFile;
|
||||
}
|
||||
@@ -205,7 +211,8 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 6> LoadOrder = {
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
@@ -370,29 +377,22 @@ namespace JSON {
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string ltrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
|
||||
std::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
}
|
||||
|
||||
return String;
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string rtrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string trim(std::string String) {
|
||||
return rtrim(ltrim(String));
|
||||
}
|
||||
|
||||
|
||||
std::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
@@ -401,7 +401,7 @@ namespace JSON {
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = trim(ManagerStr);
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
@@ -434,12 +434,27 @@ namespace JSON {
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
constexpr uint32_t MinCoreNumber = 0;
|
||||
#else
|
||||
constexpr uint32_t MinCoreNumber = 1;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
// If running the interpreter then disable cache code compilation
|
||||
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
@@ -604,7 +619,7 @@ namespace JSON {
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader();
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
@@ -650,9 +665,9 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader()
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation()} {
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
@@ -726,12 +741,16 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>();
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+85
-15
@@ -38,6 +38,24 @@
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
"TextDefault": "none",
|
||||
"Choices": [ "none", "read", "readwrite" ],
|
||||
"ArgumentHandler": "CacheObjectCodeHandler",
|
||||
"Desc": [
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
},
|
||||
"EnableAVX": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Determines whether or not we use the expanded register file for AVX or not"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
@@ -104,6 +122,13 @@
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
|
||||
]
|
||||
},
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
@@ -190,6 +215,16 @@
|
||||
"Useful for determining hot blocks of code",
|
||||
"Has some file writing overhead per JIT block"
|
||||
]
|
||||
},
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
"Also needs x86_64-linux-gnu-objdump in PATH.",
|
||||
"Can be very slow."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
@@ -201,36 +236,28 @@
|
||||
"Disables logging"
|
||||
]
|
||||
},
|
||||
"OutputSocket": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Socket to connect to",
|
||||
"eg: localhost:8087",
|
||||
"If set will override the OutputLog location"
|
||||
]
|
||||
},
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "stderr",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, <Filename>]"
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MMAN",
|
||||
"TextDefault": "mman",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
"Checks code for modification before execution.",
|
||||
"\tnone: No checks",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap",
|
||||
"\tfull: Validate code before every run (slow)"
|
||||
"\tmtrack: Page tracking based invalidation",
|
||||
"\tfull: Validate code before every run (slow)",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
|
||||
]
|
||||
},
|
||||
"TSOEnabled": {
|
||||
@@ -241,6 +268,21 @@
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -274,6 +316,16 @@
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
]
|
||||
},
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
|
||||
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
|
||||
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
|
||||
"Can be useful for Wine applications that rely on stack unwinding"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
@@ -299,6 +351,13 @@
|
||||
"Desc": [
|
||||
"Loads an AOT IR cache for the loaded executable."
|
||||
]
|
||||
},
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -316,6 +375,17 @@
|
||||
"Type": "str",
|
||||
"Default": ""
|
||||
},
|
||||
"APP_CONFIG_NAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"This is the application config name that has been loaded.",
|
||||
"This differs from APP_FILENAME in two ways",
|
||||
"Where APP_FILENAME always points to the executable path that FEX-Emu is executing.",
|
||||
"This matches what is used to load the AppLayer configuration name.",
|
||||
"When running through a compatibility layer like wine, this will only be the exe name, instead of wine full path."
|
||||
]
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
|
||||
+20
-7
@@ -43,8 +43,8 @@ namespace FEXCore::Context {
|
||||
delete CTX;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
|
||||
return CTX->InitCore(Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
return CTX->InitCore(InitialRIP, StackPointer);
|
||||
}
|
||||
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
|
||||
@@ -110,6 +110,10 @@ namespace FEXCore::Context {
|
||||
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
|
||||
}
|
||||
|
||||
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->HostFeatures;
|
||||
}
|
||||
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
CTX->HandleCallback(Thread, RIP);
|
||||
}
|
||||
@@ -149,13 +153,14 @@ namespace FEXCore::Context {
|
||||
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->CleanupAfterFork(Thread);
|
||||
}
|
||||
|
||||
|
||||
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
|
||||
CTX->SignalDelegation = SignalDelegation;
|
||||
}
|
||||
|
||||
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
|
||||
CTX->SyscallHandler = Handler;
|
||||
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
|
||||
@@ -186,11 +191,19 @@ namespace FEXCore::Context {
|
||||
CTX->WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
|
||||
return CTX->AddNamedRegion(Base, Length, Offset, Name);
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
|
||||
return CTX->LoadAOTIRCacheEntry(Name);
|
||||
}
|
||||
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
|
||||
return CTX->RemoveNamedRegion(Base, Length);
|
||||
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
|
||||
return CTX->UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(FEXCore::Context::Context *CTX, std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
CTX->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
|
||||
+83
-35
@@ -1,13 +1,16 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
@@ -34,13 +37,21 @@ class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
}
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
class InterpreterCore;
|
||||
class Dispatcher;
|
||||
}
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -67,6 +78,7 @@ namespace FEXCore::Context {
|
||||
friend class FEXCore::CPU::X86JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::CPU::InterpreterCore;
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
@@ -81,6 +93,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
@@ -97,19 +110,21 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
IntCallbackReturn InterpreterCallbackReturn;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
uint64_t ThreadID{};
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
std::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
std::mutex IdleWaitMutex;
|
||||
std::condition_variable IdleWaitCV;
|
||||
@@ -118,9 +133,13 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
@@ -135,7 +154,7 @@ namespace FEXCore::Context {
|
||||
Context();
|
||||
~Context();
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus() const;
|
||||
bool IsPaused() const { return !Running; }
|
||||
@@ -155,13 +174,33 @@ namespace FEXCore::Context {
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState
|
||||
static void RemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
RemoveCodeEntry(Frame->Thread, GuestRIP);
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
uint64_t GetThreadCount() const;
|
||||
@@ -171,21 +210,19 @@ namespace FEXCore::Context {
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
@@ -196,21 +233,9 @@ namespace FEXCore::Context {
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
* @param CompileThread Is this for the compile service or not?
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
* This is exposed because the CompileService needs to initialize compilers while copying data from
|
||||
* the paired InternalThreadState that it is compiling code for
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
@@ -233,7 +258,7 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
|
||||
/**
|
||||
* @brief Initializes the TLS data for a thread
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
@@ -269,8 +294,8 @@ namespace FEXCore::Context {
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
|
||||
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
@@ -297,8 +322,17 @@ namespace FEXCore::Context {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions);
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
|
||||
void MarkMemoryShared();
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
private:
|
||||
/**
|
||||
@@ -310,22 +344,36 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
|
||||
// Entry Cache
|
||||
uint64_t StartingRIP;
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
|
||||
|
||||
+166
-116
@@ -1,14 +1,15 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <signal.h>
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include <csignal>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
@@ -513,7 +514,8 @@ uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
//Only 32-bit pairs
|
||||
for(int i = 1; i < 10; i++) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg1 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::CCMP_MASK) == FEXCore::ArchHelpers::Arm64::CCMP_INST) {
|
||||
ExpectedReg2 = GetRmReg(NextInstr);
|
||||
@@ -535,7 +537,6 @@ uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
}
|
||||
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
@@ -1580,7 +1581,7 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
@@ -1593,7 +1594,7 @@ bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
if (Size == 2) {
|
||||
auto Res = DoLoad16(Addr);
|
||||
@@ -1623,7 +1624,7 @@ bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
@@ -1636,7 +1637,7 @@ bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
constexpr bool DoRetry = false;
|
||||
if (Size == 2) {
|
||||
@@ -1746,7 +1747,8 @@ static uint64_t HandleCAS_NoAtomics(void *_ucontext, void *_info)
|
||||
#endif
|
||||
DesiredReg = GetRdReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg = GetRmReg(NextInstr);
|
||||
}
|
||||
}
|
||||
@@ -1829,11 +1831,13 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
// Scan forward at most five instructions to find our instructions
|
||||
for (size_t i = 1; i < 6; ++i) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_SHIFT_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ADD;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_SHIFT_INST) {
|
||||
uint32_t RnReg = GetRnReg(NextInstr);
|
||||
if (RnReg == REGISTER_MASK) {
|
||||
// Zero reg means neg
|
||||
@@ -1844,21 +1848,34 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST ) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::AND_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_AND;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::BIC_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_BIC;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::OR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_OR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ORN_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ORN;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EOR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EOR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EON_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EON;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
@@ -1889,40 +1906,53 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
constexpr bool DoRetry = true;
|
||||
|
||||
auto NOPExpected = []<typename AtomicType>(AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto BICDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & ~Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto ORNDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | ~Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto EONDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ ~Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
if (Size == 2) {
|
||||
using AtomicType = uint16_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -1938,17 +1968,27 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1967,38 +2007,6 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
else if (Size == 4) {
|
||||
using AtomicType = uint32_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -2014,17 +2022,27 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2043,38 +2061,6 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
else if (Size == 8) {
|
||||
using AtomicType = uint64_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -2090,17 +2076,27 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2141,7 +2137,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
@@ -2165,7 +2161,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
@@ -2187,6 +2183,60 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t LDUR = 0b0011'1000'0100'0000'0000'0000'0000'0000;
|
||||
LDUR |= Size << 30;
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t STUR = 0b0011'1000'0000'0000'0000'0000'0000'0000;
|
||||
STUR |= Size << 30;
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXP_MASK) == FEXCore::ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
//Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleCASPAL_ARMv8(ucontext, info, Instr);
|
||||
|
||||
+22
-9
@@ -12,6 +12,10 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
|
||||
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
|
||||
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
|
||||
|
||||
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
|
||||
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
|
||||
|
||||
@@ -27,13 +31,19 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
|
||||
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
|
||||
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t EON_INST = 0x4A'20'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
@@ -46,8 +56,11 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
TYPE_ADD,
|
||||
TYPE_SUB,
|
||||
TYPE_AND,
|
||||
TYPE_BIC,
|
||||
TYPE_OR,
|
||||
TYPE_ORN,
|
||||
TYPE_EOR,
|
||||
TYPE_EON,
|
||||
TYPE_NEG, // This is just a sub with zero. Need to know the differences
|
||||
};
|
||||
|
||||
@@ -83,8 +96,8 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
|
||||
+134
-40
@@ -1,22 +1,28 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "cpu-features.h"
|
||||
#include "aarch64/instructions-aarch64.h"
|
||||
#include "utils-vixl.h"
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
#include <cpu-features.h>
|
||||
#include <utils-vixl.h>
|
||||
|
||||
#include <array>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
@@ -28,20 +34,77 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::
|
||||
SetCPUFeatures(Features);
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = Reg.IsX();
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
if (Is64Bit && ((~Constant)>> 16) == 0) {
|
||||
movn(Reg, (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
int NumMoves = 1;
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
// We only want to use ADRP+ADD if we have more than 1 segment
|
||||
for (size_t i = 0; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
if (Part != 0) {
|
||||
++RequiredMoveSegments;
|
||||
}
|
||||
}
|
||||
|
||||
// ADRP+ADD is specifically optimized in hardware
|
||||
// Check if we can use this
|
||||
auto PC = GetCursorAddress<uint64_t>();
|
||||
|
||||
// PC aligned to page
|
||||
uint64_t AlignedPC = PC & ~0xFFFULL;
|
||||
|
||||
// Offset from aligned PC
|
||||
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
}
|
||||
else {
|
||||
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
|
||||
// 21-bit signed integer here
|
||||
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
|
||||
if (vixl::IsInt21(SmallOffset)) {
|
||||
adr(Reg, SmallOffset);
|
||||
}
|
||||
else {
|
||||
// Need to use ADRP + ADD
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
add(Reg, Reg, Constant & 0xFFF);
|
||||
NumMoves = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (NOPPad) {
|
||||
for (int i = NumMoves; i < Segments; ++i) {
|
||||
nop();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -50,7 +113,7 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
MemOperand PairOffset(sp, -16, PreIndex);
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x19, x20},
|
||||
{x21, x22},
|
||||
{x23, x24},
|
||||
@@ -113,7 +176,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
MemOperand PairOffset(sp, 16, PostIndex);
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x29, x30},
|
||||
{x27, x28},
|
||||
{x25, x26},
|
||||
@@ -128,51 +191,95 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t SpillMask) {
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & SpillMask) &&
|
||||
((1U << Reg2.GetCode()) & SpillMask)) {
|
||||
if (((1U << Reg1.GetCode()) & GPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & GPRSpillMask)) {
|
||||
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & SpillMask)) {
|
||||
else if (((1U << Reg1.GetCode()) & GPRSpillMask)) {
|
||||
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & SpillMask)) {
|
||||
else if (((1U << Reg2.GetCode()) & GPRSpillMask)) {
|
||||
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t FillMask) {
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & FillMask) &&
|
||||
((1U << Reg2.GetCode()) & FillMask)) {
|
||||
if (((1U << Reg1.GetCode()) & GPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & GPRFillMask)) {
|
||||
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FillMask)) {
|
||||
else if (((1U << Reg1.GetCode()) & GPRFillMask)) {
|
||||
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FillMask)) {
|
||||
else if (((1U << Reg2.GetCode()) & GPRFillMask)) {
|
||||
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -224,23 +331,10 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
add(sp, sp, SPOffset);
|
||||
}
|
||||
|
||||
void Arm64Emitter::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,15 +1,19 @@
|
||||
#pragma once
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "platform-vixl.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -60,10 +64,17 @@ class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t SpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t FillMask = ~0U);
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
|
||||
|
||||
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
|
||||
// We can't guarantee only the lower 64bits are used so flush everything
|
||||
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
|
||||
|
||||
void PushDynamicRegsAndLR();
|
||||
void PopDynamicRegsAndLR();
|
||||
@@ -71,11 +82,8 @@ protected:
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
|
||||
void ResetStack();
|
||||
void Align16B();
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
|
||||
@@ -124,7 +124,7 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
@@ -143,7 +143,7 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Host FPR state starts at _mcontext->reserved[0];
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
Backup->FPSR = HostState->FPSR;
|
||||
Backup->FPCR = HostState->FPCR;
|
||||
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
@@ -163,7 +163,7 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
HostState->FPCR = Backup->FPCR;
|
||||
HostState->FPSR = Backup->FPSR;
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = &CodeBuffers[0];
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (ThreadState->CTX->Config.GlobalJITNaming()) {
|
||||
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
for (auto &Buffer: CodeBuffers) {
|
||||
auto start = (uintptr_t)Buffer.Ptr;
|
||||
auto end = start + Buffer.Size;
|
||||
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
+64
-11
@@ -8,11 +8,11 @@ $end_info$
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
@@ -39,6 +39,7 @@ namespace ProductNames {
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
@@ -83,7 +84,10 @@ static uint32_t CalculateNumberOfCPUs() {
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
@@ -125,7 +129,8 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFile(Data, MIDRPath, 18)) {
|
||||
uint64_t NewMIDR{};
|
||||
if (FEXCore::StrConv::Conv(&Data.at(0), &NewMIDR)) {
|
||||
std::string_view MIDRView(&Data.at(0), 18);
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
@@ -150,7 +155,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 35> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 36> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
@@ -160,6 +165,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
@@ -403,6 +409,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
@@ -413,7 +421,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(0 << 1) | // PCLMULQDQ
|
||||
(1 << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(0 << 4) | // DS-CPL
|
||||
@@ -432,7 +440,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(0 << 20) | // SSE4.2
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
@@ -442,8 +450,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(0 << 30) | // RDRAND
|
||||
(0 << 31); // Hypervisor always returns zero
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(1 << 31); // Hypervisor always returns one
|
||||
|
||||
Res.edx =
|
||||
(1 << 0) | // FPU
|
||||
@@ -644,7 +652,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // RDSEED
|
||||
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
|
||||
(1 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // Reserved
|
||||
@@ -655,7 +663,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // Reserved
|
||||
(0 << 29) | // SHA instructions
|
||||
(1 << 29) | // SHA instructions
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
@@ -809,6 +817,47 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
// Maximum supported hypervisor leafs
|
||||
// We only expose the information leaf
|
||||
//
|
||||
// Common courtesy to follow VMWare's "Hypervisor CPUID Interface proposal"
|
||||
// 4000_0000h - Information leaf. Advertising to the software which hypervisor this is
|
||||
// 4000_0001h - 4000_000Fh - Hypervisor specific leafs. FEX can use these for anything
|
||||
// 4000_0010h - 4000_00FFh - "Generic Leafs" - Try not to overwrite, other hypervisors might expect information in these
|
||||
//
|
||||
// CPUID documentation information:
|
||||
// 4000_0000h - 4FFF_FFFFh - No existing or future CPU will return information in this range
|
||||
// Reserved entirely for VMs to do whatever they want.
|
||||
Res.eax = 0x40000001;
|
||||
|
||||
// EBX, EDX, ECX become the hypervisor ID signature
|
||||
constexpr static char HypervisorID[12] = "FEXIFEXIEMU";
|
||||
memcpy(&Res.ebx, HypervisorID, sizeof(HypervisorID));
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Leaf == 0) {
|
||||
// EAX[3:0] Is the host architecture that FEX is running under
|
||||
#ifdef _M_X86_64
|
||||
// EAX[3:0] = 1 = x86_64 host architecture
|
||||
Res.eax |= 0b0001;
|
||||
#elif defined(_M_ARM_64)
|
||||
// EAX[3:0] = 2 = AArch64 host architecture
|
||||
Res.eax |= 0b0010;
|
||||
#else
|
||||
// EAX[3:0] = 0 = Unknown architecture
|
||||
#endif
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Highest extended function implemented
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
@@ -899,8 +948,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
(1 << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(0 << 30) | // 3DNow! Extensions
|
||||
(0 << 31); // 3DNow!
|
||||
(1 << 30) | // 3DNow! Extensions
|
||||
(1 << 31); // 3DNow!
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -1201,6 +1250,10 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
#ifndef CPUID_AMD
|
||||
RegisterFunction(0x1A, &CPUIDEmu::Function_1Ah);
|
||||
#endif
|
||||
// Hypervisor CPUID information leaf
|
||||
RegisterFunction(0x4000'0000, &CPUIDEmu::Function_4000_0000h);
|
||||
RegisterFunction(0x4000'0001, &CPUIDEmu::Function_4000_0001h);
|
||||
|
||||
// Largest extended function number
|
||||
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
|
||||
// Processor vendor
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -78,6 +79,8 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
|
||||
@@ -1,165 +0,0 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
#include "FEXCore/HLE/Linux/ThreadManagement.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <memory>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
|
||||
namespace FEXCore {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CompileService *This = reinterpret_cast<FEXCore::CompileService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
CompileService::CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ParentThread {Thread} {
|
||||
|
||||
CompileThreadData = std::make_unique<FEXCore::Core::InternalThreadState>();
|
||||
CompileThreadData->IsCompileService = true;
|
||||
|
||||
// We need a compiler for this work thread
|
||||
CTX->InitializeCompiler(CompileThreadData.get(), true);
|
||||
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CompileService::Initialize() {
|
||||
// Share CompileService which = this
|
||||
CompileThreadData->CompileService = ParentThread->CompileService;
|
||||
}
|
||||
|
||||
void CompileService::Shutdown() {
|
||||
ShuttingDown = true;
|
||||
// Kick the working thread
|
||||
StartWork.NotifyAll();
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
|
||||
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
|
||||
if (CompileMutex.try_lock()) {
|
||||
// We can only clear these things if we pulled the compile mutex
|
||||
|
||||
// Grab the work queue and clear it
|
||||
// We don't need to grab the queue mutex since this thread will no longer receive any work events
|
||||
// Threads are bounded 1:1
|
||||
while (!WorkQueue.empty()) {
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Go through the garbage collection array and clear it
|
||||
// It's safe to clear things that aren't marked safe since we are clearing cache
|
||||
GCArray.clear();
|
||||
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LocalIRCache.empty(), "Compile service must never have LocalIRCache");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
|
||||
// Clear the inverse cache of what is calling us from the Context ClearCache routine
|
||||
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
|
||||
SelectedThread->LookupCache->ClearCache();
|
||||
SelectedThread->CPUBackend->ClearCache();
|
||||
}
|
||||
|
||||
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
|
||||
WorkItem* ResultItem = nullptr;
|
||||
|
||||
{
|
||||
// Tell the worker thread to compile code for us
|
||||
auto Item = std::make_unique<WorkItem>();
|
||||
Item->RIP = RIP;
|
||||
|
||||
// Fill the threads work queue
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
ResultItem = WorkQueue.emplace(std::move(Item)).get();
|
||||
}
|
||||
|
||||
// Notify the thread that it has more work
|
||||
StartWork.NotifyAll();
|
||||
|
||||
return ResultItem;
|
||||
}
|
||||
|
||||
void CompileService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16]{};
|
||||
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
|
||||
while (true) {
|
||||
// Wait for work
|
||||
StartWork.Wait();
|
||||
if (ShuttingDown.load()) {
|
||||
break;
|
||||
}
|
||||
|
||||
std::scoped_lock lk(CompileMutex);
|
||||
size_t WorkItems{};
|
||||
|
||||
do {
|
||||
// Grab a work item
|
||||
std::unique_ptr<WorkItem> Item{};
|
||||
{
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
Item = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
}
|
||||
|
||||
// If we had a work item then work on it
|
||||
if (Item) {
|
||||
// Make sure it's not in lookup cache by accident
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->CurrentFrame->State.rip = Item->RIP;
|
||||
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
if (!CodePtr) {
|
||||
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
|
||||
ERROR_AND_DIE_FMT("Couldn't compile code for thread at RIP: 0x{:x}", Item->RIP);
|
||||
}
|
||||
|
||||
Item->CodePtr = CodePtr;
|
||||
Item->IRList = IRList;
|
||||
Item->DebugData = DebugData;
|
||||
Item->RAData = RAData;
|
||||
Item->StartAddr = StartAddr;
|
||||
Item->Length = Length;
|
||||
|
||||
auto& GCItem = GCArray.emplace_back(std::move(Item));
|
||||
GCItem->ServiceWorkDone.NotifyAll();
|
||||
}
|
||||
} while (WorkItems != 0);
|
||||
|
||||
// Clean up any safe entries in our GC array if we have any.
|
||||
std::erase_if(GCArray, [](const auto& Entry) {
|
||||
return Entry->SafeToClear.load(std::memory_order_relaxed);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,68 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <stdint.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
};
|
||||
class CompileService final {
|
||||
public:
|
||||
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
void Initialize();
|
||||
void Shutdown();
|
||||
|
||||
struct WorkItem {
|
||||
// Incoming
|
||||
uint64_t RIP{};
|
||||
|
||||
// Outgoing
|
||||
void *CodePtr{};
|
||||
FEXCore::IR::IRListView *IRList{};
|
||||
FEXCore::IR::RegisterAllocationData *RAData{};
|
||||
FEXCore::Core::DebugData *DebugData{};
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
|
||||
// Communication
|
||||
Event ServiceWorkDone{};
|
||||
std::atomic_bool SafeToClear{};
|
||||
};
|
||||
|
||||
WorkItem *CompileCode(uint64_t RIP);
|
||||
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address) const {
|
||||
return CompileThreadData->CPUBackend->IsAddressInJITCode(Address, false, false);
|
||||
}
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ParentThread;
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::unique_ptr<FEXCore::Core::InternalThreadState> CompileThreadData;
|
||||
|
||||
std::mutex QueueMutex{};
|
||||
std::mutex CompileMutex{};
|
||||
std::queue<std::unique_ptr<WorkItem>> WorkQueue{};
|
||||
std::vector<std::unique_ptr<WorkItem>> GCArray{};
|
||||
Event StartWork{};
|
||||
std::atomic_bool ShuttingDown{false};
|
||||
};
|
||||
}
|
||||
+484
-273
File diff suppressed because it is too large.
Load diff
+234
-108
@@ -16,20 +16,23 @@
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "code-buffer-vixl.h"
|
||||
#include "platform-vixl.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <code-buffer-vixl.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
@@ -38,12 +41,11 @@ using namespace vixl::aarch64;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SetAllowAssembler(true);
|
||||
|
||||
DispatchPtr = GetCursorAddress<CPUBackend::AsmDispatch>();
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
|
||||
// while (true) {
|
||||
// Ptr = FindBlock(RIP)
|
||||
@@ -53,15 +55,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// Ptr();
|
||||
// }
|
||||
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
Literal l_VirtualMemory {VirtualMemorySize};
|
||||
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
|
||||
Literal l_L1Ptr {Thread->LookupCache->GetL1Pointer()};
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
Literal l_CompileBlock {GetCompileBlockPtr()};
|
||||
Literal l_ExitFunctionLink {config.ExitFunctionLink};
|
||||
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -74,11 +70,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
add(x0, sp, 0);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
|
||||
@@ -95,11 +91,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify x2 since it contains our RIP once the block doesn't exist
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
|
||||
auto RipReg = x2;
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, &l_L1Ptr);
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -107,25 +103,22 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
br(x3);
|
||||
} else {
|
||||
b(&CallBlock);
|
||||
}
|
||||
br(x3);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(x0, &l_PagePtr);
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
and_(x3, RipReg, VirtualMemorySize - 1);
|
||||
}
|
||||
else {
|
||||
ldr(x3, &l_VirtualMemory);
|
||||
LoadConstant(x3, VirtualMemorySize);
|
||||
and_(x3, RipReg, x3);
|
||||
}
|
||||
|
||||
@@ -159,44 +152,21 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(x0, &l_L1Ptr);
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
|
||||
// Jump to the block
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
br(x3);
|
||||
} else {
|
||||
bind(&CallBlock);
|
||||
mov(x0, STATE);
|
||||
blr(x3);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, &l_CTX);
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
// If the value == 0 then branch to the top
|
||||
cbz(x0, &LoopTop);
|
||||
// Else we need to pause now
|
||||
b(&ThreadPauseHandler);
|
||||
} else {
|
||||
// Unconditionally loop to the top
|
||||
// We will only stop on error when compiling a block or signal
|
||||
b(&LoopTop);
|
||||
}
|
||||
}
|
||||
br(x3);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
@@ -211,7 +181,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -233,11 +203,10 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
svc(0);
|
||||
}
|
||||
|
||||
ldr(x0, &l_ExitFunctionLinkThis);
|
||||
mov(x1, STATE);
|
||||
mov(x2, lr);
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
|
||||
ldr(x3, &l_ExitFunctionLink);
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
blr(x3);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -258,7 +227,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
mov(x0, x4);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
br(x0);
|
||||
}
|
||||
@@ -267,7 +236,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
bind(&NoBlock);
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -314,7 +283,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
add(sp, sp, 16);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
b(&LoopTop);
|
||||
@@ -331,42 +300,44 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, X86State::X86_TRAPNO_OF);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)));
|
||||
LoadConstant(w1, 0x80);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)));
|
||||
LoadConstant(x1, 0);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)));
|
||||
brk(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(x1, 0);
|
||||
ldr(x1, MemOperand(x1));
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
bind(&ThreadPauseHandler);
|
||||
@@ -401,7 +372,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
|
||||
// When the thunk itself returns, it'll do its regular return logic there
|
||||
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
CallbackPtr = GetCursorAddress<CPUBackend::JITCallback>();
|
||||
CallbackPtr = GetCursorAddress<JITCallback>();
|
||||
|
||||
// We expect the thunk to have previously pushed the registers it was using
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -410,42 +381,112 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
mov(STATE, x0);
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
|
||||
ldr(w2, MemOperand(x0));
|
||||
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
add(w2, w2, 1);
|
||||
str(w2, MemOperand(x0));
|
||||
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(x2, x2, 16);
|
||||
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
str(x0, MemOperand(x2));
|
||||
|
||||
// Store RIP to the context state
|
||||
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
str(x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// load static regs
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
place(&l_VirtualMemory);
|
||||
place(&l_PagePtr);
|
||||
place(&l_L1Ptr);
|
||||
{
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
place(&l_CTX);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
place(&l_ExitFunctionLink);
|
||||
place(&l_ExitFunctionLinkThis);
|
||||
|
||||
|
||||
FinalizeCode();
|
||||
@@ -463,35 +504,120 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
|
||||
for(int i = 0; i < SRA64.size(); i++) {
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
|
||||
emit.ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
emit.ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(w0, &RunBlock);
|
||||
{
|
||||
Literal l_GuestRIP {GuestRIP};
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(x0, &l_GuestRIP);
|
||||
emit.str(x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(x0);
|
||||
emit.place(&l_GuestRIP);
|
||||
}
|
||||
emit.bind(&RunBlock);
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label InlineIRData;
|
||||
|
||||
emit.mov(x0, STATE);
|
||||
emit.adr(x1, &InlineIRData);
|
||||
|
||||
emit.ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
emit.blr(x3);
|
||||
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
emit.br(x0);
|
||||
|
||||
emit.bind(&InlineIRData);
|
||||
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
}
|
||||
|
||||
for(int i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&ThreadState->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
AArch64.LDIVHandler = LDIVHandlerAddress;
|
||||
AArch64.LUREMHandler = LUREMHandlerAddress;
|
||||
AArch64.LREMHandler = LREMHandlerAddress;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -15,10 +15,20 @@ namespace FEXCore::CPU {
|
||||
|
||||
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
public:
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
protected:
|
||||
void SpillSRA(void *ucontext, uint32_t IgnoreMask) override;
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress{};
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
};
|
||||
|
||||
}
|
||||
+150
-96
@@ -1,8 +1,8 @@
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -16,9 +16,9 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <bits/types/siginfo_t.h>
|
||||
#include <csignal>
|
||||
#include <cstring>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -40,7 +40,7 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
|
||||
// We can end up getting a signal at any point in our host state
|
||||
// Jump to a handler that saves all state so we can safely return
|
||||
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -65,7 +65,7 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
// Save guest state
|
||||
// We can't guarantee if registers are in context or host GPRs
|
||||
// So we need to save everything
|
||||
memcpy(&Context->GuestState, ThreadState->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Set the new SP
|
||||
ArchHelpers::Context::SetSp(ucontext, NewSP);
|
||||
@@ -82,13 +82,13 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
}
|
||||
|
||||
void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
|
||||
uint64_t OldSP{};
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
|
||||
OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -99,17 +99,18 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
SignalFrames.pop();
|
||||
}
|
||||
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
uintptr_t NewSP = OldSP;
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
|
||||
// First thing, reset the guest state
|
||||
memcpy(ThreadState->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Now restore host state
|
||||
ArchHelpers::Context::RestoreContext(ucontext, Context);
|
||||
|
||||
if (Context->UContextLocation) {
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
|
||||
// XXX: Unsupported since it needs state reconstruction
|
||||
@@ -133,7 +134,7 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
@@ -159,10 +160,22 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
memcpy(Frame->State.xmm, fpstate->_xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
@@ -191,7 +204,7 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
@@ -216,16 +229,26 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
@@ -274,14 +297,34 @@ static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Signal, ucontext);
|
||||
template <typename T>
|
||||
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
fpstate->sw_reserved.magic1 = x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
|
||||
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
|
||||
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_FP |
|
||||
x86_64::fpx_sw_bytes::FEATURE_SSE;
|
||||
if (is_avx_enabled) {
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_YMM;
|
||||
}
|
||||
|
||||
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
|
||||
|
||||
if (is_avx_enabled) {
|
||||
xstate->xstate_hdr.xfeatures = 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
// Set the new PC
|
||||
@@ -293,14 +336,15 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
uint64_t NewGuestSP = OldGuestSP;
|
||||
|
||||
// Pulling from context here
|
||||
bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
const bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
const uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (SRAEnabled) {
|
||||
if (IsAddressInJITCode(OldPC, false)) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
@@ -324,11 +368,11 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
#endif
|
||||
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(ucontext, IgnoreMask);
|
||||
SpillSRA(Thread, ucontext, IgnoreMask);
|
||||
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
|
||||
} else {
|
||||
if (!IsAddressInJITCode(OldPC, true)) {
|
||||
if (!IsAddressInDispatcher(OldPC)) {
|
||||
// This is likely to cause issues but in some cases it isn't fatal
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
@@ -374,8 +418,13 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
if (GuestAction->sa_flags & SA_SIGINFO) {
|
||||
// Setup ucontext a bit
|
||||
if (Is64BitMode) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::_libc_fpstate));
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86_64::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
@@ -397,8 +446,9 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
@@ -410,11 +460,12 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
@@ -443,9 +494,21 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
@@ -470,8 +533,13 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
else {
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
@@ -494,21 +562,23 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
@@ -528,15 +598,26 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
@@ -619,14 +700,14 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
else {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SignalReturn;
|
||||
LOGMAN_THROW_A_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
// This state will be restored on rt_sigreturn
|
||||
memset(Frame->State.xmm, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.xmm.avx.data, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
|
||||
Frame->State.FCW = 0x37F;
|
||||
Frame->State.FTW = 0xFFFF;
|
||||
@@ -634,44 +715,44 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
|
||||
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Signal, ucontext);
|
||||
StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
@@ -682,9 +763,9 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -695,16 +776,16 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
|
||||
|
||||
// Our ref counting doesn't matter anymore
|
||||
SignalHandlerRefCounter = 0;
|
||||
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
|
||||
|
||||
// Set the new PC
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
@@ -713,24 +794,24 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (ThreadState->RunningEvents.ThreadSleeping) {
|
||||
if (Thread->RunningEvents.ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
++ThreadState->CTX->IdleWaitRefCount;
|
||||
++Thread->CTX->IdleWaitRefCount;
|
||||
}
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -749,31 +830,4 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
for (auto iter = CodeBuffers.begin(); iter != CodeBuffers.end(); ++iter) {
|
||||
auto [start, end] = *iter;
|
||||
if (start == reinterpret_cast<uint64_t>(start_to_remove)) {
|
||||
CodeBuffers.erase(iter);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher, bool IncludeCompileService) const {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (IncludeDispatcher && IsAddressInDispatcher(Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (IncludeCompileService && ThreadState->CompileService && ThreadState->CompileService->IsAddressInJITCode(Address)) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
+48
-43
@@ -1,12 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <bits/types/stack_t.h>
|
||||
#include <cstdint>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
@@ -21,22 +19,20 @@ struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
struct DispatcherConfig {
|
||||
bool ExecuteBlocksWithCall = false;
|
||||
uintptr_t ExitFunctionLink = 0;
|
||||
uintptr_t ExitFunctionLinkThis = 0;
|
||||
bool StaticRegisterAssignment = false;
|
||||
bool StaticRegisterAllocation = false;
|
||||
};
|
||||
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
CPUBackend::AsmDispatch DispatchPtr;
|
||||
CPUBackend::JITCallback CallbackPtr;
|
||||
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
|
||||
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -48,61 +44,70 @@ public:
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
uint64_t GuestSignal_SIGILL{};
|
||||
uint64_t GuestSignal_SIGTRAP{};
|
||||
uint64_t GuestSignal_SIGSEGV{};
|
||||
uint64_t IntCallbackReturnAddress{};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
struct SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
} SynchronousFaultData;
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(int Signal, void *info, void *ucontext);
|
||||
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
|
||||
void RegisterCodeBuffer(uint8_t* start, size_t size) {
|
||||
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
|
||||
reinterpret_cast<uint64_t>(start + size));
|
||||
}
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const;
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ThreadState {Thread} {}
|
||||
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(int Signal, void *ucontext);
|
||||
void RestoreThreadState(void *ucontext);
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
static constexpr size_t MaxInterpreterTrampolineSize = 128;
|
||||
|
||||
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
|
||||
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
, config {Config}
|
||||
{}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
|
||||
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
|
||||
bool SRAEnabled = false;
|
||||
virtual void SpillSRA(void *ucontext, uint32_t IgnoreMask) {}
|
||||
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
DispatcherConfig config;
|
||||
|
||||
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
|
||||
static uint64_t GetCompileBlockPtr();
|
||||
|
||||
private:
|
||||
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
|
||||
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
};
|
||||
|
||||
}
|
||||
+213
-76
@@ -18,21 +18,26 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include "xbyak/xbyak.h"
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE r14
|
||||
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread)
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: Dispatcher(ctx, config)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
|
||||
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
|
||||
nullptr) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
|
||||
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
|
||||
DispatchPtr = getCurr<AsmDispatch>();
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -78,11 +83,10 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
|
||||
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
|
||||
|
||||
Label LoopTop;
|
||||
Label FullLookup;
|
||||
Label CallBlock;
|
||||
Label NoBlock;
|
||||
Label ExitBlock;
|
||||
Label ThreadPauseHandler;
|
||||
@@ -92,30 +96,26 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
{
|
||||
// Load our RIP
|
||||
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
|
||||
mov(rdx, qword STATE_PTR(CPUState, rip));
|
||||
|
||||
// L1 Cache
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rax, rdx);
|
||||
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
cmp(qword[r13 + rax + 8], rdx);
|
||||
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
|
||||
jne(FullLookup);
|
||||
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(qword[r13 + rax + 0]);
|
||||
} else {
|
||||
mov(rax, qword[r13 + rax + 0]);
|
||||
jmp(CallBlock);
|
||||
}
|
||||
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
|
||||
|
||||
L(FullLookup);
|
||||
mov(r13, Thread->LookupCache->GetPagePointer());
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Full lookup
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
mov(rax, rdx);
|
||||
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
mov(rbx, VirtualMemorySize - 1);
|
||||
and_(rax, rbx);
|
||||
shr(rax, 12);
|
||||
|
||||
@@ -142,8 +142,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
je(NoBlock);
|
||||
|
||||
// Update L1
|
||||
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rcx, 1);
|
||||
@@ -151,30 +150,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
mov(qword[r13 + rcx*8 + 0], rax);
|
||||
|
||||
// Real block if we made it here
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(rax);
|
||||
} else {
|
||||
L(CallBlock);
|
||||
mov(rdi, STATE);
|
||||
call(rax);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(LoopTop);
|
||||
// Else we need to pause now
|
||||
jmp(ThreadPauseHandler);
|
||||
ud2();
|
||||
}
|
||||
else {
|
||||
jmp(LoopTop);
|
||||
}
|
||||
}
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -193,10 +169,37 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
mov(rsi, STATE);
|
||||
@@ -204,20 +207,84 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
call(rax);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, config.ExitFunctionLinkThis);
|
||||
mov(rsi, STATE);
|
||||
mov(rdx, rax); // rax is set at the block end
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
mov(rax, config.ExitFunctionLink);
|
||||
call(rax);
|
||||
jmp(rax);
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
mov(rsi, rax); // rax is set at the block end
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else {
|
||||
jmp(rax);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
@@ -237,7 +304,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
}
|
||||
|
||||
{
|
||||
CallbackPtr = getCurr<CPUBackend::JITCallback>();
|
||||
CallbackPtr = getCurr<JITCallback>();
|
||||
|
||||
push(rbx);
|
||||
push(rbp);
|
||||
@@ -252,8 +319,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
// XXX: XMM?
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
|
||||
add(dword [rax], 1);
|
||||
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
@@ -261,12 +327,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
|
||||
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
|
||||
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
mov(qword [rbx], rax);
|
||||
|
||||
// Store RIP to the context state
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
|
||||
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
|
||||
|
||||
// Back to the loop top now
|
||||
jmp(LoopTop);
|
||||
@@ -281,30 +347,34 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = getCurr<uint64_t>();
|
||||
GuestSignal_SIGILL = getCurr<uint64_t>();
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = getCurr<uint64_t>();
|
||||
GuestSignal_SIGTRAP = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
int3();
|
||||
}
|
||||
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
add(byte [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)], 1);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)], X86State::X86_TRAPNO_OF);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)], 0);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)], 0x80);
|
||||
{
|
||||
// Guest SIGSEGV handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
hlt();
|
||||
}
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
IntCallbackReturnAddress = getCurr<uint64_t>();
|
||||
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
|
||||
// rdi = thread
|
||||
@@ -337,26 +407,93 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
|
||||
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
|
||||
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
emit.je(RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.mov(rax, GuestRIP);
|
||||
emit.mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
|
||||
|
||||
// Stop the thread
|
||||
emit.mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.jmp(rax);
|
||||
}
|
||||
|
||||
emit.L(RunBlock);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
|
||||
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
Label InlineIRData;
|
||||
|
||||
emit.mov(rdi, STATE);
|
||||
emit.lea(rsi, ptr[rip + InlineIRData]);
|
||||
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
|
||||
emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
|
||||
emit.L(InlineIRData);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<X86Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -17,7 +17,10 @@ namespace FEXCore::CPU {
|
||||
|
||||
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
virtual ~X86Dispatcher() override;
|
||||
};
|
||||
|
||||
+75
-47
@@ -19,6 +19,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
@@ -177,22 +178,17 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
|
||||
Decoder::Decoder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN } {
|
||||
// Using mmap is a start-up time optimization
|
||||
// Take advantage of page faulting to reduce startup time for minimal runtime cost
|
||||
DecodedBuffer =
|
||||
reinterpret_cast<FEXCore::X86Tables::DecodedInst *>(
|
||||
FEXCore::Allocator::mmap(0, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize,
|
||||
PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN }
|
||||
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
}
|
||||
|
||||
Decoder::~Decoder() {
|
||||
FEXCore::Allocator::munmap(DecodedBuffer, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize);
|
||||
PoolObject.UnclaimBuffer();
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
@@ -204,14 +200,7 @@ uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
}
|
||||
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
if (Size == 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (Size > sizeof(uint64_t)) {
|
||||
LOGMAN_MSG_A_FMT("Unknown data size to read");
|
||||
return 0;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
@@ -351,13 +340,15 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
if (Displacement) {
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
else if (ModRM.mod == 0) {
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
@@ -408,7 +399,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
|
||||
uint8_t DestSize{};
|
||||
@@ -532,7 +523,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
@@ -583,8 +574,11 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
return false;
|
||||
}
|
||||
else {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&NonGPR, ModRM);
|
||||
// Only decode if we haven't pre-decoded
|
||||
if (NonGPR.IsNone()) {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&NonGPR, ModRM);
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -638,7 +632,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
@@ -663,7 +657,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
@@ -689,7 +683,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
"REX PREFIX should have been decoded before this!");
|
||||
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
@@ -746,7 +740,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
|
||||
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -857,7 +851,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
case 0x0F: {// Escape Op
|
||||
uint8_t EscapeOp = ReadByte();
|
||||
switch (EscapeOp) {
|
||||
case 0x0F: { // 3DNow!
|
||||
case 0x0F: [[unlikely]] { // 3DNow!
|
||||
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
@@ -870,8 +864,12 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
|
||||
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&DecodeInst->Src[0], ModRM);
|
||||
// All 3DNow! instructions have the second argument as the rm handler
|
||||
// We need to decode it upfront to get the displacement out of the way
|
||||
if (ModRM.mod != 0b11) {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&DecodeInst->Src[0], ModRM);
|
||||
}
|
||||
|
||||
// Take a peek at the op just past the displacement
|
||||
uint8_t LocalOp = ReadByte();
|
||||
@@ -880,20 +878,19 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = 1;
|
||||
constexpr uint16_t PF_38_F2 = 2;
|
||||
constexpr uint16_t PF_38_F3 = 3;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
uint16_t Prefix = PF_38_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0xF2) {
|
||||
// Repeat prefix or instruction-specific
|
||||
Prefix = PF_38_F2;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF3) {
|
||||
// Repeat prefix or instruction-specific
|
||||
Prefix = PF_38_F3;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66) {
|
||||
// Operand size
|
||||
Prefix = PF_38_66;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
|
||||
Prefix |= PF_38_66;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
|
||||
Prefix |= PF_38_F2;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
|
||||
Prefix |= PF_38_F3;
|
||||
}
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
@@ -1134,7 +1131,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1142,6 +1139,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// XXX: Load symbol data
|
||||
SymbolAvailable = false;
|
||||
@@ -1163,6 +1161,13 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
// Entry is a jump target
|
||||
BlocksToDecode.emplace(PC);
|
||||
|
||||
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
|
||||
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1179,10 +1184,33 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpMinAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
|
||||
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpMaxPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMaxPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", PC + PCOffset, PC);
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
|
||||
+8
-2
@@ -27,7 +27,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
@@ -35,9 +35,14 @@ public:
|
||||
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
private:
|
||||
// To pass any information from instruction prefixes
|
||||
// down into the actual instruction handling machinery.
|
||||
@@ -63,6 +68,7 @@ private:
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
uint8_t const *InstStream;
|
||||
|
||||
+590
-433
File diff suppressed because it is too large.
Load diff
+14
-1
@@ -6,8 +6,10 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
@@ -27,6 +29,10 @@ public:
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
|
||||
void AlertLibrariesChanged() {
|
||||
LibraryMapChanged = true;
|
||||
}
|
||||
|
||||
private:
|
||||
void Break(int signal);
|
||||
|
||||
@@ -38,6 +44,9 @@ private:
|
||||
|
||||
void SendACK(std::ostream &stream, bool NACK);
|
||||
|
||||
Event ThreadBreakEvent{};
|
||||
void WaitForThreadWakeup();
|
||||
|
||||
struct HandledPacketType {
|
||||
std::string Response{};
|
||||
enum ResponseType {
|
||||
@@ -74,9 +83,13 @@ private:
|
||||
bool NoAckMode{false};
|
||||
bool NonStopMode{false};
|
||||
std::string ThreadString{};
|
||||
std::string MemoryMapString{};
|
||||
std::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
|
||||
+34
-5
@@ -1,5 +1,5 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
@@ -53,14 +53,22 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
@@ -78,6 +86,27 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef _M_X86_64
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
|
||||
+224
-219
@@ -18,16 +18,16 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 4: {
|
||||
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Pair);
|
||||
uint64_t Result{};
|
||||
Result = Src[0] & ~0U;
|
||||
Result |= Src[1] << 32;
|
||||
GD = Result;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -38,7 +38,13 @@ DEF_OP(Constant) {
|
||||
|
||||
DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
GD = Data->CurrentEntry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
GD = (Data->CurrentEntry + Op->Offset) & Mask;
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -63,11 +69,11 @@ DEF_OP(CycleCounter) {
|
||||
|
||||
DEF_OP(Add) {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a + b; };
|
||||
auto *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
auto *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a + b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(4, uint32_t, Func)
|
||||
@@ -78,11 +84,11 @@ DEF_OP(Add) {
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a - b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a - b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(4, uint32_t, Func)
|
||||
@@ -93,9 +99,9 @@ DEF_OP(Sub) {
|
||||
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = -static_cast<int32_t>(Src);
|
||||
@@ -109,10 +115,10 @@ DEF_OP(Neg) {
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -132,10 +138,10 @@ DEF_OP(Mul) {
|
||||
|
||||
DEF_OP(UMul) {
|
||||
auto Op = IROp->C<IR::IROp_UMul>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -155,9 +161,9 @@ DEF_OP(UMul) {
|
||||
|
||||
DEF_OP(Div) {
|
||||
auto Op = IROp->C<IR::IROp_Div>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -173,7 +179,7 @@ DEF_OP(Div) {
|
||||
GD = static_cast<int64_t>(Src1) / static_cast<int64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Src1) / *GetSrc<__int128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -183,10 +189,10 @@ DEF_OP(Div) {
|
||||
|
||||
DEF_OP(UDiv) {
|
||||
auto Op = IROp->C<IR::IROp_UDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -202,7 +208,7 @@ DEF_OP(UDiv) {
|
||||
GD = static_cast<uint64_t>(Src1) / static_cast<uint64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src1) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -212,10 +218,10 @@ DEF_OP(UDiv) {
|
||||
|
||||
DEF_OP(Rem) {
|
||||
auto Op = IROp->C<IR::IROp_Rem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -231,7 +237,7 @@ DEF_OP(Rem) {
|
||||
GD = static_cast<int64_t>(Src1) % static_cast<int64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Src1) % *GetSrc<__int128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -241,10 +247,10 @@ DEF_OP(Rem) {
|
||||
|
||||
DEF_OP(URem) {
|
||||
auto Op = IROp->C<IR::IROp_URem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -260,7 +266,7 @@ DEF_OP(URem) {
|
||||
GD = static_cast<uint64_t>(Src1) % static_cast<uint64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src1) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -270,10 +276,10 @@ DEF_OP(URem) {
|
||||
|
||||
DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
@@ -292,10 +298,10 @@ DEF_OP(MulH) {
|
||||
|
||||
DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint64_t>(Src1) * static_cast<uint64_t>(Src2);
|
||||
@@ -318,11 +324,11 @@ DEF_OP(UMulH) {
|
||||
|
||||
DEF_OP(Or) {
|
||||
auto Op = IROp->C<IR::IROp_Or>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a | b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a | b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -336,11 +342,11 @@ DEF_OP(Or) {
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a & b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a & b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -355,8 +361,8 @@ DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
constexpr auto Func = [](auto a, auto b) {
|
||||
using Type = decltype(a);
|
||||
return static_cast<Type>(a & static_cast<Type>(~b));
|
||||
@@ -373,11 +379,11 @@ DEF_OP(Andn) {
|
||||
|
||||
DEF_OP(Xor) {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a ^ b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a ^ b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -390,11 +396,11 @@ DEF_OP(Xor) {
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint32_t>(Src1) << (Src2 & Mask);
|
||||
@@ -408,11 +414,11 @@ DEF_OP(Lshl) {
|
||||
|
||||
DEF_OP(Lshr) {
|
||||
auto Op = IROp->C<IR::IROp_Lshr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint32_t>(Src1) >> (Src2 & Mask);
|
||||
@@ -426,11 +432,11 @@ DEF_OP(Lshr) {
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
auto Op = IROp->C<IR::IROp_Ashr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = (uint32_t)(static_cast<int32_t>(Src1) >> (Src2 & Mask));
|
||||
@@ -444,12 +450,12 @@ DEF_OP(Ashr) {
|
||||
|
||||
DEF_OP(Ror) {
|
||||
auto Op = IROp->C<IR::IROp_Ror>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Ror = [] (auto In, auto R) {
|
||||
auto RotateMask = sizeof(In) * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const auto Ror = [] (auto In, auto R) {
|
||||
const auto RotateMask = sizeof(In) * 8 - 1;
|
||||
R &= RotateMask;
|
||||
return (In >> R) | (In << (sizeof(In) * 8 - R));
|
||||
};
|
||||
@@ -468,11 +474,11 @@ DEF_OP(Ror) {
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
|
||||
__uint128_t Result{};
|
||||
Result = Src1;
|
||||
Result <<= sizeof(Src1) * 8;
|
||||
@@ -494,7 +500,7 @@ DEF_OP(Extr) {
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto Op = IROp->C<IR::IROp_PDep>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (OpSize != 4 && OpSize != 8) {
|
||||
@@ -502,10 +508,10 @@ DEF_OP(PDep) {
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Input)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Input);
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Mask)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Mask);
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Index = 0; Mask > 0; Index++) {
|
||||
@@ -526,10 +532,10 @@ DEF_OP(PExt) {
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Input)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Input);
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Mask)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Mask);
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Offset = 0; Mask > 0; Offset++) {
|
||||
@@ -543,39 +549,39 @@ DEF_OP(PExt) {
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
int32_t Res = Source / Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const int32_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
int64_t Res = Source / Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const int64_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __int128_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
@@ -587,39 +593,39 @@ DEF_OP(LDiv) {
|
||||
|
||||
DEF_OP(LUDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
uint32_t Res = Source / Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const uint32_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
uint64_t Res = Source / Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const uint64_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __uint128_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
@@ -631,39 +637,39 @@ DEF_OP(LUDiv) {
|
||||
|
||||
DEF_OP(LRem) {
|
||||
auto Op = IROp->C<IR::IROp_LRem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit Remainder from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
int32_t Res = Source % Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const int32_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
int64_t Res = Source % Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const int64_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __int128_t Res = Source % Divisor;
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
break;
|
||||
@@ -674,39 +680,39 @@ DEF_OP(LRem) {
|
||||
|
||||
DEF_OP(LURem) {
|
||||
auto Op = IROp->C<IR::IROp_LURem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit Remainder from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
uint32_t Res = Source % Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const uint32_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
uint64_t Res = Source % Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const uint64_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source % Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __uint128_t Res = Source % Divisor;
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
break;
|
||||
@@ -717,62 +723,62 @@ DEF_OP(LURem) {
|
||||
|
||||
DEF_OP(Not) {
|
||||
auto Op = IROp->C<IR::IROp_Not>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t mask[9]= { 0, 0xFF, 0xFFFF, 0, 0xFFFFFFFF, 0, 0, 0, 0xFFFFFFFFFFFFFFFFULL };
|
||||
uint64_t Mask = mask[OpSize];
|
||||
const uint64_t Mask = mask[OpSize];
|
||||
GD = (~Src) & Mask;
|
||||
}
|
||||
|
||||
DEF_OP(Popcount) {
|
||||
auto Op = IROp->C<IR::IROp_Popcount>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::popcount(Src);
|
||||
}
|
||||
|
||||
DEF_OP(FindLSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindLSB>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Result = FindFirstSetBit(Src);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t Result = FindFirstSetBit(Src);
|
||||
GD = Result - 1;
|
||||
}
|
||||
|
||||
DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FindMSB size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FindTrailingZeros) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeros>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
@@ -782,26 +788,26 @@ DEF_OP(FindTrailingZeros) {
|
||||
|
||||
DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
@@ -811,12 +817,12 @@ DEF_OP(CountLeadingZeroes) {
|
||||
|
||||
DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Src)); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Src)); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Src)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown REV size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
@@ -824,36 +830,36 @@ DEF_OP(Rev) {
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
if (Op->Width == 64) {
|
||||
SourceMask = ~0ULL;
|
||||
uint64_t DestMask = ~(SourceMask << Op->lsb);
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
|
||||
}
|
||||
const uint64_t DestMask = ~(SourceMask << Op->lsb);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Dest);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 8, "OpSize is too large for BFE: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
if (Op->Width == 64) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
SourceMask <<= Op->lsb;
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = (Src & SourceMask) >> Op->lsb;
|
||||
}
|
||||
|
||||
DEF_OP(Sbfe) {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 8, "OpSize is too large for SBFE: {}", OpSize);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for SBFE: {}", IROp->Size);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
const uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
Src <<= ShiftLeftAmount;
|
||||
Src >>= ShiftRightAmount;
|
||||
GD = Src;
|
||||
@@ -861,20 +867,20 @@ DEF_OP(Sbfe) {
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
uint64_t ArgTrue;
|
||||
uint64_t ArgFalse;
|
||||
|
||||
if (OpSize == 4) {
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[3]);
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->FalseVal);
|
||||
} else {
|
||||
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[3]);
|
||||
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->FalseVal);
|
||||
}
|
||||
|
||||
bool CompResult;
|
||||
@@ -889,30 +895,29 @@ DEF_OP(Select) {
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Header.Args[0]);
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 16, "OpSize is too large for VExtractToGPR: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Idx * 8;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Idx * 8;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
@@ -921,25 +926,25 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
DEF_OP(Float_ToGPR_ZS) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
@@ -948,25 +953,25 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
|
||||
DEF_OP(Float_ToGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
@@ -977,9 +982,9 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
uint32_t ResultFlags{};
|
||||
if (Op->ElementSize == 4) {
|
||||
float Src1 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
float Src2 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
const float Src1 = *GetSrc<float*>(Data->SSAData, Op->Scalar1);
|
||||
const float Src2 = *GetSrc<float*>(Data->SSAData, Op->Scalar2);
|
||||
const bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
if (Unordered || (Src1 < Src2)) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
@@ -997,9 +1002,9 @@ DEF_OP(FCmp) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Scalar1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Scalar2);
|
||||
const bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
if (Unordered || (Src1 < Src2)) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
|
||||
+112
-113
@@ -313,30 +313,29 @@ uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *add
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Size is the size of each pair element
|
||||
switch (OpSize) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Header.Args[2]);
|
||||
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Addr);
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Expected);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Desired);
|
||||
|
||||
__uint128_t Expected = Src1;
|
||||
bool Result = MemData->compare_exchange_strong(Expected, Src2);
|
||||
memcpy(GDP, Result ? &Src1 : &Expected, 16);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", IROp->ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -347,33 +346,33 @@ DEF_OP(CAS) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint8_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint8_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint16_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint16_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint32_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint32_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
@@ -385,26 +384,26 @@ DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
@@ -416,26 +415,26 @@ DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
@@ -447,26 +446,26 @@ DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
@@ -478,26 +477,26 @@ DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
@@ -509,26 +508,26 @@ DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
@@ -540,29 +539,29 @@ DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
@@ -575,29 +574,29 @@ DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
@@ -610,29 +609,29 @@ DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
@@ -645,29 +644,29 @@ DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
@@ -680,29 +679,29 @@ DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
@@ -715,29 +714,29 @@ DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
@@ -751,22 +750,22 @@ DEF_OP(AtomicFetchNeg) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
using Type = uint8_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
using Type = uint16_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
using Type = uint32_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
using Type = uint64_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -25,24 +26,13 @@ static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestReturn) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
SignalReturn(Data->State);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
Data->State->CTX->InterpreterCallbackReturn(Data->State, Data->StackEntry);
|
||||
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
|
||||
}
|
||||
|
||||
DEF_OP(ExitFunction) {
|
||||
@@ -52,7 +42,7 @@ DEF_OP(ExitFunction) {
|
||||
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
|
||||
|
||||
void *ContextData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->NewRIP);
|
||||
|
||||
memcpy(ContextData, Src, OpSize);
|
||||
|
||||
@@ -61,22 +51,22 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->Header.Args[0]);
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TargetBlock);
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
bool CompResult;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
@@ -137,7 +127,7 @@ DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->Header.Args[0]));
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
@@ -151,15 +141,15 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
Data->State->CTX->RemoveCodeEntry(Data->State, Data->CurrentEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
|
||||
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
|
||||
|
||||
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
|
||||
@@ -14,12 +14,12 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
uint64_t Offset = Op->Index * Op->Header.ElementSize * 8;
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
@@ -35,31 +35,31 @@ DEF_OP(VInsGPR) {
|
||||
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), Op->Header.ElementSize);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
@@ -68,15 +68,15 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 8);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 4);
|
||||
break;
|
||||
}
|
||||
@@ -86,14 +86,14 @@ DEF_OP(Float_FToF) {
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
@@ -104,14 +104,14 @@ DEF_OP(Vector_SToF) {
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -122,14 +122,14 @@ DEF_OP(Vector_FToZS) {
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -140,14 +140,14 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- float
|
||||
// Only the lower elements from the source
|
||||
@@ -172,17 +172,17 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
const auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
const auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
|
||||
@@ -298,12 +298,69 @@ namespace AES {
|
||||
}
|
||||
}
|
||||
|
||||
namespace CRC32 {
|
||||
// CRC32 per byte lookup table.
|
||||
constexpr std::array<uint32_t, 256> CRC32CTable = []() consteval {
|
||||
std::array<uint32_t, 256> Table{};
|
||||
|
||||
// Clang 11.x doesn't support bitreverse as a consteval
|
||||
// constexpr uint32_t Polynomial = 0x1EDC6F41;
|
||||
constexpr uint32_t PolynomialRev = 0x82F63B78; //__builtin_bitreverse32(Polynomial);
|
||||
|
||||
for (size_t Char = 0; Char < std::size(Table); ++Char) {
|
||||
uint32_t CurrentChar = Char;
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
if (CurrentChar & 1) {
|
||||
CurrentChar = (CurrentChar >> 1) ^ PolynomialRev;
|
||||
}
|
||||
else {
|
||||
CurrentChar >>= 1;
|
||||
}
|
||||
}
|
||||
Table[Char] = CurrentChar;
|
||||
}
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
uint32_t crc32cb(uint32_t Accumulator, uint8_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ data] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32ch(uint32_t Accumulator, uint16_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cw(uint32_t Accumulator, uint32_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cx(uint32_t Accumulator, uint64_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 32) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 40) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 48) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 56) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
|
||||
// Pseudo-code
|
||||
// Dst = InvMixColumns(STATE)
|
||||
@@ -314,8 +371,8 @@ DEF_OP(AESImc) {
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -334,8 +391,8 @@ DEF_OP(AESEnc) {
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -352,8 +409,8 @@ DEF_OP(AESEncLast) {
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -372,8 +429,8 @@ DEF_OP(AESDec) {
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -390,7 +447,7 @@ DEF_OP(AESDecLast) {
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
// Pseudo-code
|
||||
// X3 = Src1[127:96]
|
||||
@@ -429,6 +486,71 @@ DEF_OP(AESKeyGenAssist) {
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
uint32_t Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Src1);
|
||||
uint8_t *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->Src2);
|
||||
uint32_t Tmp{};
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
Tmp = CRC32::crc32cb(Src1, *(uint8_t*)Src2);
|
||||
break;
|
||||
case 2:
|
||||
Tmp = CRC32::crc32ch(Src1, *(uint16_t*)Src2);
|
||||
break;
|
||||
case 4:
|
||||
Tmp = CRC32::crc32cw(Src1, *(uint32_t*)Src2);
|
||||
break;
|
||||
case 8:
|
||||
Tmp = CRC32::crc32cx(Src1, *(uint64_t*)Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown CRC32C size: {}", Op->SrcSize);
|
||||
break;
|
||||
|
||||
}
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
const auto Selector = Op->Selector;
|
||||
auto* Dst = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
auto* Src1 = GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
auto* Src2 = GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
const uint64_t TMP1 = (Selector & 0x01) == 0 ? Src1[0] : Src1[1];
|
||||
const uint64_t TMP2 = (Selector & 0x10) == 0 ? Src2[0] : Src2[1];
|
||||
|
||||
const auto make_lo = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 0; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs << i;
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
const auto make_hi = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 1; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs >> (64 - i);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
|
||||
Dst[0] = make_lo(TMP1, TMP2);
|
||||
Dst[1] = make_hi(TMP1, TMP2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+138
-76
@@ -20,99 +20,90 @@ DEF_OP(F80LOADFCW) {
|
||||
|
||||
DEF_OP(F80ADD) {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SUB) {
|
||||
auto Op = IROp->C<IR::IROp_F80Sub>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80MUL) {
|
||||
auto Op = IROp->C<IR::IROp_F80Mul>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80DIV) {
|
||||
auto Op = IROp->C<IR::IROp_F80Div>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F80FYL2X>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80ATAN>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM1>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F80SCALE>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CVT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVT>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
@@ -131,9 +122,9 @@ DEF_OP(F80CVT) {
|
||||
|
||||
DEF_OP(F80CVTINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
@@ -158,116 +149,112 @@ DEF_OP(F80CVTINT) {
|
||||
DEF_OP(F80CVTTO) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->Header.Args[0]);
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTTOINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80ROUND) {
|
||||
auto Op = IROp->C<IR::IROp_F80Round>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80F2XM1>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::F2XM1(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::F2XM1(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80TAN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FTAN(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FTAN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SQRT) {
|
||||
auto Op = IROp->C<IR::IROp_F80SQRT>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSQRT(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSQRT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F80SIN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSIN(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSIN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80COS) {
|
||||
auto Op = IROp->C<IR::IROp_F80COS>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FCOS(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FCOS(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_EXP) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_SIG) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CMP) {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
uint32_t ResultFlags{};
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
bool eq, lt, nan;
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
@@ -288,7 +275,7 @@ DEF_OP(F80CMP) {
|
||||
|
||||
DEF_OP(F80BCDLOAD) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
@@ -323,7 +310,7 @@ DEF_OP(F80BCDLOAD) {
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
@@ -356,6 +343,81 @@ DEF_OP(F80BCDSTORE) {
|
||||
memcpy(GDP, BCD, 10);
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = sin(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = cos(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = tan(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64F2XM1>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = exp2(Src) - 1.0;
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64ATAN>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = atan2(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = fmod(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = remainder(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F64FYL2X>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = Src2 * log2(Src1);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double trunc = (double)(int64_t)(Src2); //truncate
|
||||
const double Tmp = Src1 * exp2(trunc);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -222,11 +222,80 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(double src) {
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(double src) {
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(double src) {
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(double src) {
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(double src1, double src2) {
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(double src1, double src2) {
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(double src1, double src2) {
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(double src1, double src2) {
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(double src1, double src2) {
|
||||
double trunc = (double)(int64_t)(src2); //truncate
|
||||
return src1 * exp2(trunc);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(Src1);
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]) >> Op->Flag) & 1;
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Value) >> Op->Flag) & 1;
|
||||
}
|
||||
#undef DEF_OP
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Dispatcher;
|
||||
class X86DispatchGenerator;
|
||||
class Arm64DispatchGenerator;
|
||||
|
||||
@@ -20,28 +21,27 @@ using DestMapType = std::vector<uint32_t>;
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
explicit InterpreterCore(Dispatcher *Dispatch,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *State;
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher{};
|
||||
size_t BufferUsed;
|
||||
Dispatcher *Dispatch;
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
|
||||
@@ -9,67 +9,101 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <memory>
|
||||
#include <bits/types/stack_t.h>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
|
||||
#include "InterpreterOps.h"
|
||||
|
||||
#if defined(_M_X86_64)
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#elif defined(_M_ARM_64)
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#else
|
||||
#error missing arch
|
||||
#endif
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Dispatch(Dispatcher)
|
||||
{
|
||||
|
||||
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
|
||||
InterpreterOps::InterpretIR(Thread, Thread->CurrentFrame->State.rip, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
|
||||
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: CTX {ctx}
|
||||
, State {Thread} {
|
||||
|
||||
if (!CompileThread &&
|
||||
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
CreateAsmDispatch(ctx, Thread);
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
|
||||
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
|
||||
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
|
||||
|
||||
auto DestBuffer = BufferStart;
|
||||
|
||||
if (GDBEnabled) {
|
||||
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
|
||||
DestBuffer += GDBSize;
|
||||
BufferUsed += GDBSize;
|
||||
}
|
||||
|
||||
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
|
||||
DestBuffer += TrampolineSize;
|
||||
BufferUsed += TrampolineSize;
|
||||
|
||||
|
||||
IR->Serialize(DestBuffer);
|
||||
DestBuffer += IRSize;
|
||||
BufferUsed += IRSize;
|
||||
|
||||
return BufferStart;
|
||||
}
|
||||
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
void InterpreterCore::ClearCache() {
|
||||
// Calling this one is needed to setup the initial CurrentCodeBuffer
|
||||
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
BufferUsed = 0;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<InterpreterCore>(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
InterpreterCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
}
|
||||
@@ -12,9 +12,11 @@ namespace FEXCore::Core {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
struct DispatcherConfig;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -159,21 +159,26 @@
|
||||
break; \
|
||||
}
|
||||
|
||||
struct InterpVector256 {
|
||||
__uint128_t Lower;
|
||||
__uint128_t Upper;
|
||||
};
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.ID().Value];
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.Value];
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
|
||||
template<typename Res>
|
||||
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Src.ID().Value];
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Src.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
@@ -0,0 +1,313 @@
|
||||
#include "FEXCore/Core/CoreState.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/F80Ops.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_UNKNOWN, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F32, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I32, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F32_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
|
||||
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
|
||||
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8).fn);
|
||||
Info[Core::OPINDEX_F80CMP_0] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, Core::OPINDEX_F80CMP_0).fn);
|
||||
Info[Core::OPINDEX_F80CMP_1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>, Core::OPINDEX_F80CMP_1).fn);
|
||||
Info[Core::OPINDEX_F80CMP_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, Core::OPINDEX_F80CMP_2).fn);
|
||||
Info[Core::OPINDEX_F80CMP_3] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>, Core::OPINDEX_F80CMP_3).fn);
|
||||
Info[Core::OPINDEX_F80CMP_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, Core::OPINDEX_F80CMP_4).fn);
|
||||
Info[Core::OPINDEX_F80CMP_5] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>, Core::OPINDEX_F80CMP_5).fn);
|
||||
Info[Core::OPINDEX_F80CMP_6] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, Core::OPINDEX_F80CMP_6).fn);
|
||||
Info[Core::OPINDEX_F80CMP_7] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>, Core::OPINDEX_F80CMP_7).fn);
|
||||
Info[Core::OPINDEX_F80CVTTOINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2).fn);
|
||||
Info[Core::OPINDEX_F80CVTTOINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4).fn);
|
||||
|
||||
// Unary
|
||||
Info[Core::OPINDEX_F80ROUND] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ROUND>::handle, Core::OPINDEX_F80ROUND).fn);
|
||||
Info[Core::OPINDEX_F80F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80F2XM1>::handle, Core::OPINDEX_F80F2XM1).fn);
|
||||
Info[Core::OPINDEX_F80TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80TAN>::handle, Core::OPINDEX_F80TAN).fn);
|
||||
Info[Core::OPINDEX_F80SQRT] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SQRT>::handle, Core::OPINDEX_F80SQRT).fn);
|
||||
Info[Core::OPINDEX_F80SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SIN>::handle, Core::OPINDEX_F80SIN).fn);
|
||||
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle, Core::OPINDEX_F80COS).fn);
|
||||
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle, Core::OPINDEX_F80XTRACT_EXP).fn);
|
||||
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle, Core::OPINDEX_F80XTRACT_SIG).fn);
|
||||
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle, Core::OPINDEX_F80BCDSTORE).fn);
|
||||
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle, Core::OPINDEX_F80BCDLOAD).fn);
|
||||
|
||||
// Binary
|
||||
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle, Core::OPINDEX_F80ADD).fn);
|
||||
Info[Core::OPINDEX_F80SUB] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SUB>::handle, Core::OPINDEX_F80SUB).fn);
|
||||
Info[Core::OPINDEX_F80MUL] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80MUL>::handle, Core::OPINDEX_F80MUL).fn);
|
||||
Info[Core::OPINDEX_F80DIV] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle, Core::OPINDEX_F80DIV).fn);
|
||||
Info[Core::OPINDEX_F80FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle, Core::OPINDEX_F80FYL2X).fn);
|
||||
Info[Core::OPINDEX_F80ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle, Core::OPINDEX_F80ATAN).fn);
|
||||
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
|
||||
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
|
||||
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
|
||||
|
||||
// Double Precision
|
||||
Info[Core::OPINDEX_F64SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle, Core::OPINDEX_F64SIN).fn);
|
||||
Info[Core::OPINDEX_F64COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle, Core::OPINDEX_F64COS).fn);
|
||||
Info[Core::OPINDEX_F64TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle, Core::OPINDEX_F64TAN).fn);
|
||||
Info[Core::OPINDEX_F64ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle, Core::OPINDEX_F64ATAN).fn);
|
||||
Info[Core::OPINDEX_F64F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle, Core::OPINDEX_F64F2XM1).fn);
|
||||
Info[Core::OPINDEX_F64FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle, Core::OPINDEX_F64FYL2X).fn);
|
||||
Info[Core::OPINDEX_F64FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle, Core::OPINDEX_F64FPREM).fn);
|
||||
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
|
||||
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
if (Op->Truncate) {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2);
|
||||
}
|
||||
else {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
if (Op->Truncate) {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4);
|
||||
}
|
||||
else {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
if (Op->Truncate) {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8);
|
||||
}
|
||||
else {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags));
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
// Double Precision Unary
|
||||
COMMON_F64_OP(F2XM1)
|
||||
COMMON_F64_OP(TAN)
|
||||
COMMON_F64_OP(SIN)
|
||||
COMMON_F64_OP(COS)
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_F64_OP(FYL2X)
|
||||
COMMON_F64_OP(ATAN)
|
||||
COMMON_F64_OP(FPREM1)
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "InterpreterDefines.h"
|
||||
#include "InterpreterOps.h"
|
||||
#include "F80Ops.h"
|
||||
|
||||
@@ -112,9 +113,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(GUESTRETURN, GuestReturn);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -124,7 +122,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
|
||||
// Conversion ops
|
||||
@@ -167,6 +165,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, NoOp);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -176,6 +175,8 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
@@ -185,8 +186,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(CREATEVECTOR2, CreateVector2);
|
||||
REGISTER_OP(CREATEVECTOR4, CreateVector4);
|
||||
REGISTER_OP(SPLATVECTOR2, SplatVector);
|
||||
REGISTER_OP(SPLATVECTOR4, SplatVector);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
@@ -275,6 +274,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
@@ -283,6 +283,8 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// F80 ops
|
||||
REGISTER_OP(F80LOADFCW, F80LOADFCW);
|
||||
@@ -311,6 +313,17 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(F80BCDLOAD, F80BCDLOAD);
|
||||
REGISTER_OP(F80BCDSTORE, F80BCDSTORE);
|
||||
|
||||
// F64 ops
|
||||
REGISTER_OP(F64SIN, F64SIN);
|
||||
REGISTER_OP(F64COS, F64COS);
|
||||
REGISTER_OP(F64TAN, F64TAN);
|
||||
REGISTER_OP(F64F2XM1, F64F2XM1);
|
||||
REGISTER_OP(F64ATAN, F64ATAN);
|
||||
REGISTER_OP(F64FPREM, F64FPREM);
|
||||
REGISTER_OP(F64FPREM1, F64FPREM1);
|
||||
REGISTER_OP(F64FYL2X, F64FYL2X);
|
||||
REGISTER_OP(F64SCALE, F64SCALE);
|
||||
|
||||
return Handlers;
|
||||
}();
|
||||
|
||||
@@ -321,237 +334,37 @@ void InterpreterOps::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, IROpData *Data
|
||||
void InterpreterOps::Op_NoOp(FEXCore::IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...)) {
|
||||
return {FABI_UNKNOWN, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float)) {
|
||||
return {FABI_F80_F32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double)) {
|
||||
return {FABI_F80_F64, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t)) {
|
||||
return {FABI_F80_I16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t)) {
|
||||
return {FABI_VOID_U16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t)) {
|
||||
return {FABI_F80_I32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I16_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_I64_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_F80_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags]);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData) {
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *CurrentIR) {
|
||||
volatile void *StackEntry = alloca(0);
|
||||
|
||||
// Debug data is only passed in debug builds
|
||||
#ifndef NDEBUG
|
||||
// TODO: should be moved to an IR Op
|
||||
Thread->Stats.InstructionsExecuted.fetch_add(DebugData->GuestInstructionCount);
|
||||
#endif
|
||||
|
||||
uintptr_t ListSize = CurrentIR->GetSSACount();
|
||||
const uintptr_t ListSize = CurrentIR->GetSSACount();
|
||||
|
||||
static_assert(sizeof(FEXCore::IR::IROp_Header) == 4);
|
||||
static_assert(sizeof(FEXCore::IR::OrderedNode) == 16);
|
||||
|
||||
auto BlockEnd = CurrentIR->GetBlocks().end();
|
||||
|
||||
InterpreterOps::IROpData OpData{};
|
||||
OpData.State = Thread;
|
||||
OpData.SSAData = alloca(ListSize * 16);
|
||||
OpData.CurrentEntry = Entry;
|
||||
OpData.CurrentIR = CurrentIR;
|
||||
OpData.StackEntry = StackEntry;
|
||||
OpData.BlockIterator = CurrentIR->GetBlocks().begin();
|
||||
constexpr size_t ListEntrySizeInBytes = sizeof(InterpVector256);
|
||||
const size_t SSADataSize = ListSize * ListEntrySizeInBytes;
|
||||
|
||||
// Clear them all to zero. Required for Zero-extend semantics
|
||||
memset(OpData.SSAData, 0, ListSize * 16);
|
||||
InterpreterOps::IROpData OpData{
|
||||
.State = Frame->Thread,
|
||||
.CurrentEntry = Frame->State.rip,
|
||||
.CurrentIR = CurrentIR,
|
||||
.StackEntry = StackEntry,
|
||||
.SSAData = alloca(SSADataSize),
|
||||
.BlockResults = {},
|
||||
.BlockIterator = CurrentIR->GetBlocks().begin(),
|
||||
};
|
||||
|
||||
// Clear all SSAData entries to zero. Required for Zero-extend semantics
|
||||
memset(OpData.SSAData, 0, SSADataSize);
|
||||
|
||||
while (1) {
|
||||
using namespace FEXCore::IR;
|
||||
auto [BlockNode, BlockHeader] = OpData.BlockIterator();
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
// Reset the block results per block
|
||||
memset(&OpData.BlockResults, 0, sizeof(OpData.BlockResults));
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
@@ -27,6 +28,8 @@ namespace FEXCore::CPU {
|
||||
FABI_F80_I32,
|
||||
FABI_F32_F80,
|
||||
FABI_F64_F80,
|
||||
FABI_F64_F64,
|
||||
FABI_F64_F64_F64,
|
||||
FABI_I16_F80,
|
||||
FABI_I32_F80,
|
||||
FABI_I64_F80,
|
||||
@@ -38,18 +41,20 @@ namespace FEXCore::CPU {
|
||||
struct FallbackInfo {
|
||||
FallbackABI ABI;
|
||||
void *fn;
|
||||
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
|
||||
};
|
||||
|
||||
class InterpreterOps {
|
||||
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
uint64_t CurrentEntry{};
|
||||
FEXCore::IR::IRListView *CurrentIR{};
|
||||
FEXCore::IR::IRListView const *CurrentIR{};
|
||||
volatile void *StackEntry{};
|
||||
void *SSAData{};
|
||||
struct {
|
||||
@@ -137,9 +142,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(GuestReturn);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -149,7 +151,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -194,6 +196,8 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -203,8 +207,6 @@ namespace FEXCore::CPU {
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(CreateVector2);
|
||||
DEF_OP(CreateVector4);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
@@ -290,6 +292,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
@@ -298,6 +301,8 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
|
||||
///< F80 ops
|
||||
DEF_OP(F80LOADFCW);
|
||||
@@ -325,6 +330,17 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(F80CMP);
|
||||
DEF_OP(F80BCDLOAD);
|
||||
DEF_OP(F80BCDSTORE);
|
||||
|
||||
//< F64 ops
|
||||
DEF_OP(F64SIN);
|
||||
DEF_OP(F64COS);
|
||||
DEF_OP(F64TAN);
|
||||
DEF_OP(F64F2XM1);
|
||||
DEF_OP(F64ATAN);
|
||||
DEF_OP(F64FPREM);
|
||||
DEF_OP(F64FPREM1);
|
||||
DEF_OP(F64FYL2X);
|
||||
DEF_OP(F64SCALE);
|
||||
#undef DEF_OP
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
@@ -391,7 +407,7 @@ namespace FEXCore::CPU {
|
||||
return CompResult;
|
||||
}
|
||||
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
|
||||
return IROp->Size;
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -58,7 +59,7 @@ DEF_OP(StoreContext) {
|
||||
ContextPtr += Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
@@ -72,7 +73,7 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
@@ -102,14 +103,14 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
}
|
||||
|
||||
@@ -133,7 +134,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
@@ -227,9 +228,9 @@ DEF_OP(StoreMem) {
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Header.Args[0]);
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[1]), 16);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
@@ -237,8 +238,8 @@ DEF_OP(VLoadMemElement) {
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Header.Args[0]); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Header.Args[1])[Op->Index], sizeof(y)); \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
|
||||
+40
-12
@@ -4,6 +4,8 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -14,6 +16,7 @@ $end_info$
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#endif
|
||||
#include <sys/random.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
@@ -43,14 +46,26 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Halt: // HLT
|
||||
StopThread(Data->State);
|
||||
|
||||
Data->State->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = 1;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.Signal = Op->Reason.Signal;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.TrapNo = Op->Reason.TrapNumber;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.err_code = Op->Reason.ErrorRegister;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.si_code = Op->Reason.si_code;
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
break;
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
case SIGTRAP:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGSEGV);
|
||||
break;
|
||||
default:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break Reason: {}", Op->Reason); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -85,7 +100,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
uint8_t GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->RoundMode);
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t HostRounding{};
|
||||
__asm volatile(R"(
|
||||
@@ -125,16 +140,16 @@ DEF_OP(SetRoundingMode) {
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (OpSize <= 8) {
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
|
||||
}
|
||||
else if (OpSize == 16) {
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src0 = Src;
|
||||
uint64_t Src1 = Src >> 64;
|
||||
const auto Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Value);
|
||||
const uint64_t Src0 = Src;
|
||||
const uint64_t Src1 = Src >> 64;
|
||||
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
|
||||
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
|
||||
}
|
||||
@@ -148,6 +163,19 @@ DEF_OP(ProcessorID) {
|
||||
GD = (CPUNode << 12) | CPU;
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
// We are ignoring Op->GetReseeded in the interpreter
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
ssize_t Result = ::getrandom(&DstPtr[0], 8, 0);
|
||||
|
||||
// Second result is if we managed to read a valid random number or not
|
||||
DstPtr[1] = Result == 8 ? 1 : 0;
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
// Nop implementation
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -14,27 +14,27 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
uintptr_t Src = GetSrc<uintptr_t>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = GetSrc<uintptr_t>(Data->SSAData, Op->Pair);
|
||||
memcpy(GDP,
|
||||
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Lower);
|
||||
const void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Upper);
|
||||
|
||||
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
|
||||
|
||||
memcpy(Dst, Src_Lower, Op->Header.Size);
|
||||
memcpy(Dst + Op->Header.Size, Src_Upper, Op->Header.Size);
|
||||
memcpy(Dst, Src_Lower, IROp->ElementSize);
|
||||
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), OpSize);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+424
-403
File diff suppressed because it is too large.
Load diff
+361
-273
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,131 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|arm64
|
||||
desc: relocation logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
+100
-103
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -12,18 +13,17 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
// Size is the size of each pair element
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
auto Expected = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
auto Expected = GetSrcPair<RA_64>(Op->Expected.ID());
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Desired.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP3, Expected.first);
|
||||
mov(TMP4, Expected.second);
|
||||
|
||||
switch (OpSize) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4:
|
||||
caspal(TMP3.W(), TMP4.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc));
|
||||
mov(Dst.first.W(), TMP3.W());
|
||||
@@ -34,11 +34,11 @@ DEF_OP(CASPair) {
|
||||
mov(Dst.first, TMP3);
|
||||
mov(Dst.second, TMP4);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
aarch64::Label LoopTop;
|
||||
aarch64::Label LoopNotExpected;
|
||||
@@ -91,7 +91,7 @@ DEF_OP(CASPair) {
|
||||
bind(&LoopExpected);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -99,16 +99,13 @@ DEF_OP(CASPair) {
|
||||
DEF_OP(CAS) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
// Args[0]: Expected
|
||||
// Args[1]: Desired
|
||||
// Args[2]: Pointer
|
||||
// DataSrc = *Src1
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
|
||||
auto Expected = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Desired = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
auto Expected = GetReg<RA_64>(Op->Expected.ID());
|
||||
auto Desired = GetReg<RA_64>(Op->Desired.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, Expected);
|
||||
@@ -216,14 +213,14 @@ DEF_OP(CAS) {
|
||||
DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 1: staddlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -234,7 +231,7 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -243,7 +240,7 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -252,7 +249,7 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -261,7 +258,7 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
add(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
add(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
@@ -274,10 +271,10 @@ DEF_OP(AtomicAdd) {
|
||||
DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
@@ -293,7 +290,7 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -302,7 +299,7 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -311,7 +308,7 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -320,7 +317,7 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
sub(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
sub(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
@@ -333,10 +330,10 @@ DEF_OP(AtomicSub) {
|
||||
DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 2: stclrlh(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
@@ -352,7 +349,7 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -361,7 +358,7 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -370,7 +367,7 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -379,7 +376,7 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
and_(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
@@ -392,14 +389,14 @@ DEF_OP(AtomicAnd) {
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: stsetl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: stsetl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 1: stsetlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 2: stsetlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 4: stsetl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 8: stsetl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -410,7 +407,7 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -419,7 +416,7 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -428,7 +425,7 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -437,7 +434,7 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
orr(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
orr(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
@@ -450,14 +447,14 @@ DEF_OP(AtomicOr) {
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: steorl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: steorl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 1: steorlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 2: steorlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 4: steorl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 8: steorl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -468,7 +465,7 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -477,7 +474,7 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -486,7 +483,7 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -495,7 +492,7 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
eor(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
eor(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
@@ -508,15 +505,15 @@ DEF_OP(AtomicXor) {
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
case 1: swpalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swpalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -527,7 +524,7 @@ DEF_OP(AtomicSwap) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
|
||||
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
uxtb(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
@@ -536,7 +533,7 @@ DEF_OP(AtomicSwap) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
|
||||
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
uxtw(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
@@ -545,7 +542,7 @@ DEF_OP(AtomicSwap) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
stlxr(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
|
||||
stlxr(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
@@ -554,7 +551,7 @@ DEF_OP(AtomicSwap) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
stlxr(TMP4, GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
|
||||
stlxr(TMP4, GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2.X());
|
||||
break;
|
||||
@@ -566,14 +563,14 @@ DEF_OP(AtomicSwap) {
|
||||
|
||||
DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
case 1: ldaddalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -584,7 +581,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -594,7 +591,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -604,7 +601,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -614,7 +611,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
add(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
add(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
@@ -627,10 +624,10 @@ DEF_OP(AtomicFetchAdd) {
|
||||
|
||||
DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -646,7 +643,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -656,7 +653,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -666,7 +663,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -676,7 +673,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
sub(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
sub(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
@@ -689,10 +686,10 @@ DEF_OP(AtomicFetchSub) {
|
||||
|
||||
DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldclralh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -708,7 +705,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -718,7 +715,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -728,7 +725,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -738,7 +735,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
and_(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
@@ -751,14 +748,14 @@ DEF_OP(AtomicFetchAnd) {
|
||||
|
||||
DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldsetal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldsetal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
case 1: ldsetalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldsetalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldsetal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldsetal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -769,7 +766,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -779,7 +776,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -789,7 +786,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -799,7 +796,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
orr(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
orr(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
@@ -812,14 +809,14 @@ DEF_OP(AtomicFetchOr) {
|
||||
|
||||
DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldeoral(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldeoral(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
case 1: ldeoralb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldeoralh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldeoral(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldeoral(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
@@ -830,7 +827,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -840,7 +837,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -850,7 +847,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -860,7 +857,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
eor(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
eor(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
@@ -873,7 +870,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
|
||||
DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
|
||||
+60
-63
@@ -4,6 +4,8 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -18,17 +20,6 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestReturn) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -36,7 +27,7 @@ DEF_OP(SignalReturn) {
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
LoadConstant(x0, ThreadSharedData.SignalReturnInstruction);
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)));
|
||||
br(x0);
|
||||
}
|
||||
|
||||
@@ -49,10 +40,10 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
|
||||
ldr(w2, MemOperand(x0));
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(x0));
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
@@ -76,7 +67,7 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress};
|
||||
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
|
||||
ldr(x0, &l_BranchHost);
|
||||
@@ -85,10 +76,10 @@ DEF_OP(ExitFunction) {
|
||||
place(&l_BranchHost);
|
||||
place(&l_BranchGuest);
|
||||
} else {
|
||||
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RipReg = GetReg<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
LoadConstant(x0, ThreadState->LookupCache->GetL1Pointer());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -99,7 +90,7 @@ DEF_OP(ExitFunction) {
|
||||
br(x1);
|
||||
|
||||
bind(&FullLookup);
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress);
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)));
|
||||
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
br(TMP1);
|
||||
}
|
||||
@@ -107,9 +98,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
@@ -184,8 +175,16 @@ DEF_OP(Syscall) {
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
SpillStaticRegs();
|
||||
}
|
||||
else {
|
||||
// Need to spill all caller saved registers still
|
||||
SpillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
|
||||
}
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(sp, sp, SPOffset);
|
||||
@@ -194,22 +193,30 @@ DEF_OP(Syscall) {
|
||||
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
|
||||
}
|
||||
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(CTX->SyscallHandler));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
|
||||
LoadConstant(x3, reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall));
|
||||
blr(x3);
|
||||
|
||||
add(sp, sp, SPOffset);
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs();
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY &&
|
||||
(Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
else {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
|
||||
}
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Move result to its destination register
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Move result to its destination register
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
@@ -340,21 +347,23 @@ DEF_OP(InlineSyscall) {
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
else {
|
||||
uxtw(GetReg<RA_64>(Node), x0);
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
else {
|
||||
uxtw(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -368,7 +377,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x0, GetReg<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
@@ -427,7 +436,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
@@ -437,7 +446,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Entry);
|
||||
|
||||
LoadConstant(x2, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
SpillStaticRegs();
|
||||
blr(x2);
|
||||
FillStaticRegs();
|
||||
@@ -454,19 +463,10 @@ DEF_OP(CPUID) {
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&CTX->CPUID));
|
||||
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
|
||||
using ClassPtrType = FEXCore::CPUID::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t, uint32_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
PtrCast Ptr;
|
||||
Ptr.ClassPtr = &FEXCore::CPUIDEmu::RunFunction;
|
||||
LoadConstant(x3, Ptr.Data);
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
@@ -483,9 +483,6 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(GUESTRETURN, GuestReturn);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -495,7 +492,7 @@ void Arm64JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -13,22 +13,22 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->Index, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
@@ -39,18 +39,18 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()).W());
|
||||
break;
|
||||
case 8:
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()).X());
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()).X());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -58,22 +58,22 @@ DEF_OP(VCastFromGPR) {
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -81,14 +81,14 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Header.Args[0].ID()).S());
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Scalar.ID()).S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Header.Args[0].ID()).D());
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Scalar.ID()).D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
@@ -99,10 +99,10 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -112,10 +112,10 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -125,11 +125,11 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
@@ -142,11 +142,11 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2S());
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
@@ -159,50 +159,50 @@ DEF_OP(Vector_FToI) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -14,57 +14,56 @@ using namespace vixl::aarch64;
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Vector.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesmc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesimc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Label Constant;
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label PastConstant;
|
||||
|
||||
// Do a "regular" AESE step
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Src.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
adr(TMP1.X(), &Constant);
|
||||
ldr(VTMP3, MemOperand(TMP1.X()));
|
||||
ldr(VTMP3, &ConstantLiteral);
|
||||
|
||||
// Now EOR in the RCON
|
||||
if (Op->RCON) {
|
||||
@@ -80,24 +79,68 @@ DEF_OP(AESKeyGenAssist) {
|
||||
}
|
||||
|
||||
b(&PastConstant);
|
||||
bind(&Constant);
|
||||
dc32(0x0B0E0104);
|
||||
dc32(0x040B0E01);
|
||||
dc32(0x0306090C);
|
||||
dc32(0x0C030609);
|
||||
place(&ConstantLiteral);
|
||||
bind(&PastConstant);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32cb(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 2:
|
||||
crc32ch(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 4:
|
||||
crc32cw(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_64>(Op->Src2.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node).Q();
|
||||
auto Src1 = GetSrc(Op->Src1.ID()).V2D();
|
||||
auto Src2 = GetSrc(Op->Src2.ID()).V2D();
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
pmull(Dst, Src1, Src2);
|
||||
break;
|
||||
case 0b00000001:
|
||||
mov(VTMP1.V1D(), Src1, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src2);
|
||||
break;
|
||||
case 0b00010000:
|
||||
mov(VTMP1.V1D(), Src2, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src1);
|
||||
break;
|
||||
case 0b00010001:
|
||||
pmull2(Dst, Src1, Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -13,7 +13,7 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+263
-241
@@ -21,10 +21,14 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
@@ -32,13 +36,46 @@ $end_info$
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
namespace {
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LDIV(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static uint64_t LUREM(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
@@ -56,8 +93,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -72,8 +108,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
LoadConstant(x0, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -92,8 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
LoadConstant(x0, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -118,8 +152,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
else {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -140,8 +173,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -160,8 +192,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -172,6 +203,43 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
@@ -180,8 +248,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -199,8 +266,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -218,8 +284,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -240,8 +305,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -259,8 +323,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -283,8 +346,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -300,59 +362,72 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
FEXCore::IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
FEXCore::Allocator::mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
{
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
}
|
||||
|
||||
// Can't allocate a code buffer until after dispatcher is created
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
*GetBuffer() = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
SetAllowAssembler(true);
|
||||
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, HostSupportsSVE{ctx->HostFeatures.SupportsAVX}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -393,95 +468,76 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
if (!CompileThread) {
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalReturnInstruction = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
|
||||
if (!Core->Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Core->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
SetAllowAssembler(true);
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::Arm64JITCore::";
|
||||
auto Buffer = GetBuffer();
|
||||
Buffer->EmitString(JITString);
|
||||
Buffer->Align();
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
auto Buffer = GetBuffer();
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
Buffer->Reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
InitialCodeBuffer.Size *= 1.5;
|
||||
InitialCodeBuffer.Size = std::min(InitialCodeBuffer.Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(InitialCodeBuffer.Size);
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = Arm64JITCore::AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
*Buffer = vixl::CodeBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
|
||||
@@ -496,12 +552,12 @@ template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg].W();
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg].W();
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -511,12 +567,12 @@ template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -537,12 +593,12 @@ std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JI
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -551,12 +607,12 @@ aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -582,7 +638,12 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = Entry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
@@ -607,13 +668,18 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x0, Entry);
|
||||
@@ -622,9 +688,9 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
// AAPCS64
|
||||
@@ -647,31 +713,11 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
auto GuestEntry = GetCursorAddress<uint64_t>();
|
||||
GuestEntry = GetCursorAddress<uint8_t *>();
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Thread))); // Get thread
|
||||
ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
cbz(w0, &RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
LoadConstant(x0, Entry);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
|
||||
// Stop the thread
|
||||
LoadConstant(x0, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
|
||||
br(x0);
|
||||
}
|
||||
bind(&RunBlock);
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
GetBuffer()->CursorForward(GDBSize);
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -693,9 +739,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
@@ -710,10 +757,6 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
bind(&IsTarget->second);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({GetCursorAddress<uintptr_t>(), 0, IR->GetID(BlockNode)});
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
|
||||
@@ -723,7 +766,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.back().HostCodeSize = GetCursorAddress<uintptr_t>() - DebugData->Subblocks.back().HostCodeStart;
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -736,68 +782,44 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = GetCursorAddress<uint64_t>();
|
||||
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(GuestEntry), CodeEnd - reinterpret_cast<uint64_t>(GuestEntry));
|
||||
auto CodeEnd = GetCursorAddress<uint8_t *>();
|
||||
CPU.EnsureIAndDCacheCoherency(GuestEntry, CodeEnd - GuestEntry);
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->HostCodeSize = CodeEnd - GuestEntry;
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
this->IR = nullptr;
|
||||
|
||||
return reinterpret_cast<void*>(GuestEntry);
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
Arm64JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
+88
-64
@@ -6,17 +6,23 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
@@ -37,14 +43,8 @@ using namespace vixl::aarch64;
|
||||
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -52,7 +52,7 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -60,21 +60,16 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
}
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
const bool HostSupportsSVE{};
|
||||
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
Label *PendingTargetLabel;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
|
||||
@@ -145,46 +140,77 @@ private:
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
// We don't want to mvoe above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
#if DEBUG
|
||||
vixl::aarch64::Disassembler Disasm;
|
||||
#endif
|
||||
|
||||
static uint64_t ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalReturnInstruction{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
@@ -275,9 +301,6 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(GuestReturn);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -287,7 +310,7 @@ private:
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -305,7 +328,7 @@ private:
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
@@ -327,7 +350,7 @@ private:
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -336,6 +359,8 @@ private:
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -345,8 +370,6 @@ private:
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(CreateVector2);
|
||||
DEF_OP(CreateVector4);
|
||||
DEF_OP(SplatVector2);
|
||||
DEF_OP(SplatVector4);
|
||||
DEF_OP(VMov);
|
||||
@@ -434,6 +457,7 @@ private:
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
@@ -442,9 +466,9 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+140
-76
@@ -4,6 +4,8 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
@@ -58,26 +60,26 @@ DEF_OP(LoadContext) {
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
strb(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 2:
|
||||
strh(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
strh(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 4:
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
str(Src.B(), MemOperand(STATE, Op->Offset));
|
||||
@@ -104,7 +106,7 @@ DEF_OP(LoadRegister) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0])) / 8;
|
||||
auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
@@ -113,30 +115,32 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_32>(Node), reg.W());
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_64>(Node), reg);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
|
||||
@@ -145,17 +149,17 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(host.B(), guest.B());
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
fmov(host.H(), guest.H());
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
fmov(host.S(), guest.S());
|
||||
@@ -165,7 +169,7 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.D(), guest.D());
|
||||
@@ -175,13 +179,13 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.Q(), guest.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -189,7 +193,7 @@ DEF_OP(StoreRegister) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = Op->Offset / 8 - 1;
|
||||
auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
@@ -198,29 +202,31 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 32);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Op->Value.ID()).GetCode() != reg.GetCode())
|
||||
mov(reg, GetReg<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
|
||||
@@ -233,36 +239,36 @@ DEF_OP(StoreRegister) {
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
ins(guest.V8H(), regOffs/2, host.V8H(), 0);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
ins(guest.V4S(), regOffs/4, host.V4S(), 0);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
ins(guest.V2D(), regOffs / 8, host.V2D(), 0);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (guest.GetCode() != host.GetCode())
|
||||
mov(guest.Q(), host.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -349,11 +355,11 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto value = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto value = GetReg<RA_64>(Op->Value.ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -392,7 +398,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Header.Args[0].ID());
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -441,25 +447,25 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * 16;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
strh(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
strh(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -467,15 +473,15 @@ DEF_OP(SpillRegister) {
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()).S(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()).S(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()).D(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()).D(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -539,7 +545,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
}
|
||||
|
||||
MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
@@ -570,7 +576,7 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -617,13 +623,43 @@ DEF_OP(LoadMem) {
|
||||
DEF_OP(LoadMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("LoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "LoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldapurb(Dst, MemSrc);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
@@ -698,29 +734,29 @@ DEF_OP(LoadMemTSO) {
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
strb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 2:
|
||||
strh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
strh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
str(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
str(Src.B(), MemSrc);
|
||||
@@ -744,28 +780,56 @@ DEF_OP(StoreMem) {
|
||||
|
||||
DEF_OP(StoreMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("StoreMemTSO: No offset allowed");
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "StoreMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlurb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlurh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlr(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlur(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlur(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlr(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
@@ -774,7 +838,7 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
else {
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
str(Src.B(), MemSrc);
|
||||
@@ -800,7 +864,7 @@ DEF_OP(StoreMemTSO) {
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
|
||||
@@ -857,7 +921,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
|
||||
DEF_OP(ParanoidStoreMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
|
||||
@@ -866,25 +930,25 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlrh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlr(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlr(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
stlr(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
mov(TMP1.W(), Src.V16B(), 0);
|
||||
@@ -934,7 +998,7 @@ DEF_OP(VStoreMemElement) {
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
@@ -949,7 +1013,7 @@ DEF_OP(CacheLineClear) {
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
|
||||
+70
-50
@@ -4,21 +4,21 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <syscall.h>
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, GetCursorAddress<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
@@ -37,44 +37,38 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
hlt(4);
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
ResetStack();
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
|
||||
br(TMP1);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
add(sp, TMP1, 0);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddressSpillSRA);
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_Interrupt3: { // INT3
|
||||
ResetStack();
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
ResetStack();
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
|
||||
br(TMP1);
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)));
|
||||
br(TMP1);
|
||||
break;
|
||||
default:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -106,7 +100,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetReg<RA_64>(Op->RoundMode.ID());
|
||||
|
||||
// Setup the rounding flags correctly
|
||||
and_(TMP1, Src, 0b11);
|
||||
@@ -141,15 +135,15 @@ DEF_OP(Print) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
LoadConstant(x3, reinterpret_cast<uint64_t>(PrintValue));
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)));
|
||||
}
|
||||
else {
|
||||
fmov(x0, GetSrc(Op->Header.Args[0].ID()).V1D());
|
||||
fmov(x0, GetSrc(Op->Value.ID()).V1D());
|
||||
// Bug in vixl that source vector needs to b V1D rather than V2D?
|
||||
fmov(x1, GetSrc(Op->Header.Args[0].ID()).V1D(), 1);
|
||||
LoadConstant(x3, reinterpret_cast<uint64_t>(PrintVectorValue));
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -210,6 +204,28 @@ DEF_OP(ProcessorID) {
|
||||
orr(GetReg<RA_64>(Node), x0, Operand(x1, LSL, 12));
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
mrs(Dst.first, RNDRRS);
|
||||
}
|
||||
else {
|
||||
mrs(Dst.first, RNDR);
|
||||
}
|
||||
|
||||
// If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100
|
||||
cset(Dst.second, Condition::ne);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
hint(SystemHint::YIELD);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -218,6 +234,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -227,6 +244,9 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,13 +15,13 @@ DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
@@ -37,18 +37,18 @@ DEF_OP(CreateElementPair) {
|
||||
aarch64::Register RegSecond;
|
||||
aarch64::Register RegTmp;
|
||||
|
||||
switch (Op->Header.Size) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetReg<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetReg<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Upper.ID());
|
||||
RegTmp = w0;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetReg<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Upper.ID());
|
||||
RegTmp = x0;
|
||||
break;
|
||||
}
|
||||
@@ -70,7 +70,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+595
-467
File diff suppressed because it is too large.
Load diff
+7
-4
@@ -14,10 +14,13 @@ namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures();
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+245
-241
File diff suppressed because it is too large.
Load diff
+67
-74
@@ -18,20 +18,16 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Args[0]: Desired
|
||||
// Args[1]: Expected
|
||||
// Args[2]: Pointer
|
||||
// DataSrc = *Src1
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
// Third operand must be a calculated guest memory address
|
||||
//OrderedNode *CASResult = _CAS(Src3, Src2, Src1);
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
auto Expected = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetSrc<RA_64>(Op->Header.Args[2].ID());
|
||||
auto Expected = GetSrcPair<RA_64>(Op->Expected.ID());
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Desired.ID());
|
||||
auto MemSrc = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
Xbyak::Reg MemReg = MemSrc;
|
||||
|
||||
@@ -47,7 +43,7 @@ DEF_OP(CASPair) {
|
||||
|
||||
lock();
|
||||
|
||||
switch (OpSize) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
cmpxchg8b(dword [MemReg]);
|
||||
// EDX:EAX now contains the result
|
||||
@@ -62,7 +58,7 @@ DEF_OP(CASPair) {
|
||||
mov(Dst.second, rdx);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -70,18 +66,15 @@ DEF_OP(CAS) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Args[0]: Desired
|
||||
// Args[1]: Expected
|
||||
// Args[2]: Pointer
|
||||
// DataSrc = *Src1
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
// Third operand must be a calculated guest memory address
|
||||
//OrderedNode *CASResult = _CAS(Src3, Src2, Src1);
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[2].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rax, GetSrc<RA_64>(Op->Expected.ID()));
|
||||
|
||||
// RCX now contains pointer
|
||||
// RAX contains our expected value
|
||||
@@ -90,23 +83,23 @@ DEF_OP(CAS) {
|
||||
lock();
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
cmpxchg(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
cmpxchg(byte [MemReg], GetSrc<RA_8>(Op->Desired.ID()));
|
||||
movzx(GetDst<RA_64>(Node), al);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
cmpxchg(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
cmpxchg(word [MemReg], GetSrc<RA_16>(Op->Desired.ID()));
|
||||
movzx(GetDst<RA_64>(Node), ax);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
cmpxchg(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
cmpxchg(dword [MemReg], GetSrc<RA_32>(Op->Desired.ID()));
|
||||
// RAX now contains the result
|
||||
mov (GetDst<RA_64>(Node), eax);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
cmpxchg(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
cmpxchg(qword [MemReg], GetSrc<RA_64>(Op->Desired.ID()));
|
||||
// RAX now contains the result
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
break;
|
||||
@@ -118,21 +111,21 @@ DEF_OP(CAS) {
|
||||
DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
lock();
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
add(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
add(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
case 2:
|
||||
add(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
add(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
case 4:
|
||||
add(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
add(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
add(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
add(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
@@ -141,20 +134,20 @@ DEF_OP(AtomicAdd) {
|
||||
DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
lock();
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
sub(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
sub(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
case 2:
|
||||
sub(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
sub(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
case 4:
|
||||
sub(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
sub(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
sub(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
sub(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
@@ -163,20 +156,20 @@ DEF_OP(AtomicSub) {
|
||||
DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
lock();
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
and_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
and_(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
case 2:
|
||||
and_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
and_(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
case 4:
|
||||
and_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
and_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
@@ -185,20 +178,20 @@ DEF_OP(AtomicAnd) {
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
lock();
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
or_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
or_(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
case 2:
|
||||
or_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
or_(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
case 4:
|
||||
or_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
or_(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
or_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
or_(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
@@ -207,20 +200,20 @@ DEF_OP(AtomicOr) {
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
lock();
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
xor_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
xor_(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
case 2:
|
||||
xor_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
xor_(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
case 4:
|
||||
xor_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
xor_(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
xor_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
xor_(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
@@ -230,26 +223,26 @@ DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
|
||||
Xbyak::Reg MemReg = rax;
|
||||
mov(MemReg, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(MemReg, GetSrc<RA_64>(Op->Addr.ID()));
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
movzx(GetDst<RA_64>(Node), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
movzx(GetDst<RA_64>(Node), GetSrc<RA_8>(Op->Value.ID()));
|
||||
lock();
|
||||
xchg(byte [MemReg], GetDst<RA_8>(Node));
|
||||
break;
|
||||
case 2:
|
||||
movzx(GetDst<RA_64>(Node), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
movzx(GetDst<RA_64>(Node), GetSrc<RA_16>(Op->Value.ID()));
|
||||
lock();
|
||||
xchg(word [MemReg], GetDst<RA_16>(Node));
|
||||
break;
|
||||
case 4:
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_32>(Op->Value.ID()));
|
||||
lock();
|
||||
xchg(dword [MemReg], GetDst<RA_32>(Node));
|
||||
break;
|
||||
case 8:
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
lock();
|
||||
xchg(qword [MemReg], GetDst<RA_64>(Node));
|
||||
break;
|
||||
@@ -260,28 +253,28 @@ DEF_OP(AtomicSwap) {
|
||||
DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
movzx(rcx, GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
movzx(rcx, GetSrc<RA_8>(Op->Value.ID()));
|
||||
lock();
|
||||
xadd(byte [MemReg], cl);
|
||||
movzx(GetDst<RA_32>(Node), cl);
|
||||
break;
|
||||
case 2:
|
||||
movzx(rcx, GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
movzx(rcx, GetSrc<RA_16>(Op->Value.ID()));
|
||||
lock();
|
||||
xadd(word [MemReg], cx);
|
||||
movzx(GetDst<RA_32>(Node), cx);
|
||||
break;
|
||||
case 4:
|
||||
mov(ecx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(ecx, GetSrc<RA_32>(Op->Value.ID()));
|
||||
lock();
|
||||
xadd(dword [MemReg], ecx);
|
||||
mov(GetDst<RA_64>(Node), ecx);
|
||||
break;
|
||||
case 8:
|
||||
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(rcx, GetSrc<RA_64>(Op->Value.ID()));
|
||||
lock();
|
||||
xadd(qword [MemReg], rcx);
|
||||
mov(GetDst<RA_64>(Node), rcx);
|
||||
@@ -293,31 +286,31 @@ DEF_OP(AtomicFetchAdd) {
|
||||
DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
mov(cl, GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
mov(cl, GetSrc<RA_8>(Op->Value.ID()));
|
||||
neg(cl);
|
||||
lock();
|
||||
xadd(byte [MemReg], cl);
|
||||
movzx(GetDst<RA_32>(Node), cl);
|
||||
break;
|
||||
case 2:
|
||||
mov(cx, GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
mov(cx, GetSrc<RA_16>(Op->Value.ID()));
|
||||
neg(cx);
|
||||
lock();
|
||||
xadd(word [MemReg], cx);
|
||||
movzx(GetDst<RA_32>(Node), cx);
|
||||
break;
|
||||
case 4:
|
||||
mov(ecx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(ecx, GetSrc<RA_32>(Op->Value.ID()));
|
||||
neg(ecx);
|
||||
lock();
|
||||
xadd(dword [MemReg], ecx);
|
||||
mov(GetDst<RA_32>(Node), ecx);
|
||||
break;
|
||||
case 8:
|
||||
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(rcx, GetSrc<RA_64>(Op->Value.ID()));
|
||||
neg(rcx);
|
||||
lock();
|
||||
xadd(qword [MemReg], rcx);
|
||||
@@ -331,7 +324,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
|
||||
// TMP1 = rax
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
@@ -341,7 +334,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt8(), TMP1.cvt8());
|
||||
mov(TMP3.cvt8(), TMP1.cvt8());
|
||||
and_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
|
||||
@@ -357,7 +350,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt16(), TMP1.cvt16());
|
||||
mov(TMP3.cvt16(), TMP1.cvt16());
|
||||
and_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
|
||||
@@ -374,7 +367,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt32(), TMP1.cvt32());
|
||||
mov(TMP3.cvt32(), TMP1.cvt32());
|
||||
and_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
|
||||
@@ -391,7 +384,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt64(), TMP1.cvt64());
|
||||
mov(TMP3.cvt64(), TMP1.cvt64());
|
||||
and_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
|
||||
@@ -409,7 +402,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
|
||||
// TMP1 = rax
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
@@ -418,7 +411,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt8(), TMP1.cvt8());
|
||||
mov(TMP3.cvt8(), TMP1.cvt8());
|
||||
or_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
|
||||
@@ -434,7 +427,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt16(), TMP1.cvt16());
|
||||
mov(TMP3.cvt16(), TMP1.cvt16());
|
||||
or_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
|
||||
@@ -451,7 +444,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt32(), TMP1.cvt32());
|
||||
mov(TMP3.cvt32(), TMP1.cvt32());
|
||||
or_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
|
||||
@@ -468,7 +461,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt64(), TMP1.cvt64());
|
||||
mov(TMP3.cvt64(), TMP1.cvt64());
|
||||
or_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
|
||||
@@ -486,7 +479,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
|
||||
// TMP1 = rax
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
@@ -495,7 +488,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt8(), TMP1.cvt8());
|
||||
mov(TMP3.cvt8(), TMP1.cvt8());
|
||||
xor_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
|
||||
@@ -511,7 +504,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt16(), TMP1.cvt16());
|
||||
mov(TMP3.cvt16(), TMP1.cvt16());
|
||||
xor_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
|
||||
@@ -528,7 +521,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt32(), TMP1.cvt32());
|
||||
mov(TMP3.cvt32(), TMP1.cvt32());
|
||||
xor_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
|
||||
@@ -545,7 +538,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt64(), TMP1.cvt64());
|
||||
mov(TMP3.cvt64(), TMP1.cvt64());
|
||||
xor_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
|
||||
@@ -562,7 +555,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
|
||||
+19
-48
@@ -29,17 +29,6 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestReturn) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
@@ -47,8 +36,7 @@ DEF_OP(SignalReturn) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
}
|
||||
|
||||
mov(TMP1, ThreadSharedData.SignalHandlerReturnAddress);
|
||||
jmp(TMP1);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
@@ -58,8 +46,7 @@ DEF_OP(CallbackReturn) {
|
||||
}
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
mov(rax, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
|
||||
sub(dword [rax], 1);
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)], 1);
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 8);
|
||||
@@ -97,14 +84,16 @@ DEF_OP(ExitFunction) {
|
||||
jmp(qword[rax]);
|
||||
|
||||
L(l_BranchHost);
|
||||
dq(ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress);
|
||||
//FEX_TODO(this is not per thread)
|
||||
dq(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
L(l_BranchGuest);
|
||||
dq(NewRIP);
|
||||
} else {
|
||||
Xbyak::Reg RipReg = GetSrc<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
mov(rcx, ThreadState->LookupCache->GetL1Pointer());
|
||||
mov(rcx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)]);
|
||||
|
||||
mov(rax, RipReg);
|
||||
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
@@ -117,9 +106,8 @@ DEF_OP(ExitFunction) {
|
||||
jmp(qword[LookupBase + 0]);
|
||||
|
||||
L(FullLookup);
|
||||
mov(rax, ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], RipReg);
|
||||
jmp(rax);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)]);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
@@ -129,9 +117,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
@@ -187,15 +175,13 @@ DEF_OP(Syscall) {
|
||||
}
|
||||
|
||||
mov(rsi, STATE); // Move thread in to rsi
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX->SyscallHandler));
|
||||
mov(rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)]);
|
||||
mov(rdx, rsp);
|
||||
|
||||
mov(rax, reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall));
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
// {rdi, rsi, rdx}
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -223,7 +209,7 @@ DEF_OP(Thunk) {
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
mov(rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rdi, GetSrc<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
|
||||
@@ -267,7 +253,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -280,9 +266,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
mov(rax, Entry); // imm64 move
|
||||
mov(rsi, rax);
|
||||
|
||||
|
||||
mov(rax, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -294,13 +278,6 @@ DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
using ClassPtrType = FEXCore::CPUID::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function, uint32_t Leaf);
|
||||
union {
|
||||
ClassPtrType ClassPtr;
|
||||
uint64_t Raw;
|
||||
} Ptr;
|
||||
Ptr.ClassPtr = &CPUIDEmu::RunFunction;
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
|
||||
@@ -311,20 +288,17 @@ DEF_OP(CPUID) {
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (edx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov (rdi, reinterpret_cast<uint64_t>(&CTX->CPUID));
|
||||
mov (edx, GetSrc<RA_32>(Op->Leaf.ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
mov(rax, Ptr.Raw);
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -340,9 +314,6 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(GUESTRETURN, GuestReturn);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -351,7 +322,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -18,23 +18,23 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
movapd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()), Op->Index);
|
||||
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()), Op->Index);
|
||||
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()), Op->Index);
|
||||
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[1].ID()), Op->Index);
|
||||
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
@@ -45,18 +45,18 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
movzx(rax, GetSrc<RA_8>(Op->Header.Args[0].ID()));
|
||||
movzx(rax, GetSrc<RA_8>(Op->Src.ID()));
|
||||
vmovq(GetDst(Node), rax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(rax, GetSrc<RA_16>(Op->Header.Args[0].ID()));
|
||||
movzx(rax, GetSrc<RA_16>(Op->Src.ID()));
|
||||
vmovq(GetDst(Node), rax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()).cvt32());
|
||||
vmovd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()).cvt32());
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()).cvt64());
|
||||
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()).cvt64());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown VCastFromGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -64,22 +64,23 @@ DEF_OP(VCastFromGPR) {
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -87,14 +88,15 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtss2sd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtss2sd(GetDst(Node), GetSrc(Op->Scalar.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtsd2ss(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtsd2ss(GetDst(Node), GetSrc(Op->Scalar.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Float_FToF sizes: 0x{:x}", Conv);
|
||||
@@ -105,7 +107,7 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
@@ -113,8 +115,8 @@ DEF_OP(Vector_SToF) {
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
@@ -127,10 +129,10 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -140,10 +142,10 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -151,15 +153,15 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
@@ -190,10 +192,10 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,44 +17,95 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
vaesimc(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
vaesimc(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
vaesenc(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesenc(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
vaesdec(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesdec(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->RCON);
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Src.ID()), Op->RCON);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (IROp->Size) {
|
||||
case 4:
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Src2.ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Src1.ID()));
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Src2.ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Src1.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", IROp->Size);
|
||||
}
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt8());
|
||||
break;
|
||||
case 2:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt16());
|
||||
break;
|
||||
case 4:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt32());
|
||||
break;
|
||||
case 8:
|
||||
crc32(GetDst<RA_64>(Node).cvt64(), TMP1.cvt64());
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node);
|
||||
auto Src1 = GetSrc(Op->Src1.ID());
|
||||
auto Src2 = GetSrc(Op->Src2.ID());
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
case 0b00000001:
|
||||
case 0b00010000:
|
||||
case 0b00010001:
|
||||
vpclmulqdq(Dst, Src1, Src2, Op->Selector);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -18,7 +18,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
shr(rax, Op->Flag);
|
||||
and_(rax, 1);
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
|
||||
+156
-193
@@ -15,6 +15,8 @@ $end_info$
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
@@ -23,11 +25,11 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bits/types/stack_t.h>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
@@ -42,33 +44,21 @@ $end_info$
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
|
||||
namespace {
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
FEXCore::Allocator::mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
void X86JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
}
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
@@ -109,9 +99,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_VOID_U16: {
|
||||
PushRegs();
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
break;
|
||||
@@ -120,9 +108,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
movss(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -136,9 +122,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -153,9 +137,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -171,9 +153,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -187,9 +167,34 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
call(rax);
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
movsd(xmm1, GetSrc(IROp->Args[1].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -203,9 +208,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -218,9 +221,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -233,9 +234,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -251,9 +250,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -266,9 +263,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -286,9 +281,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -301,23 +294,42 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
|
||||
record[0] = HostCode;
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread)
|
||||
: CodeGenerator(Buffer.Size, Buffer.Ptr, nullptr)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread}
|
||||
, InitialCodeBuffer {Buffer}
|
||||
{
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, CodeGenerator(0, this, nullptr) // this is not used here
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -346,98 +358,58 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
if (!CompileThread) {
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
{
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
ThreadSharedData.OverflowExceptionInstructionAddress = Dispatcher->OverflowExceptionInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
X86JITCore::~X86JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
|
||||
}
|
||||
|
||||
void X86JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::X86JITCore::";
|
||||
for (char c : JITString) {
|
||||
db(c);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
void X86JITCore::ClearCache() {
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(CTX, CurrentCodeBuffer->Size);
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(CTX, X86JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
setNewBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
setNewBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
LOGMAN_THROW_AA_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
}
|
||||
@@ -553,7 +525,12 @@ bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, u
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = Entry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
@@ -594,42 +571,30 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((getSize() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
void *GuestEntry = getCurr<void*>();
|
||||
GuestEntry = getCurr<uint8_t*>();
|
||||
CursorEntry = getSize();
|
||||
this->IR = IR;
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(RunBlock);
|
||||
// Else we need to pause now
|
||||
mov(rax, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
|
||||
jmp(rax);
|
||||
ud2();
|
||||
|
||||
L(RunBlock);
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
setSize(getSize() + GDBSize);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(RAData != nullptr, "Needs RA");
|
||||
LOGMAN_THROW_AA_FMT(RAData != nullptr, "Needs RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -684,12 +649,13 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
{
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = getCurr<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
|
||||
@@ -744,6 +710,13 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(getCurr<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
@@ -760,32 +733,22 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(GuestExit) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
|
||||
record[0] = HostCode;
|
||||
return HostCode;
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
X86JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
}
|
||||
+87
-53
@@ -6,8 +6,10 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
@@ -24,13 +26,6 @@ using namespace Xbyak;
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -57,9 +52,7 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
|
||||
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
CodeBuffer Buffer,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~X86JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -67,7 +60,7 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -75,20 +68,75 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
}
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
void LoadConstantWithPadding(Xbyak::Reg Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Label Offset;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(Xbyak::Reg Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(Xbyak::Reg Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint64_t CursorEntry{};
|
||||
/** @} */
|
||||
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
uint64_t Entry;
|
||||
|
||||
std::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
@@ -133,6 +181,10 @@ private:
|
||||
[[nodiscard]] Xbyak::Xmm GetSrc(IR::NodeID Node) const;
|
||||
[[nodiscard]] Xbyak::Xmm GetDst(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] static Xbyak::Ymm ToYMM(const Xbyak::Xmm& xmm) {
|
||||
return Xbyak::Ymm{xmm.getIdx()};
|
||||
}
|
||||
|
||||
[[nodiscard]] Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale) const;
|
||||
|
||||
@@ -141,41 +193,23 @@ private:
|
||||
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
FEXCore::IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
bool GetSamplingData {true};
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
static uint64_t ExitFunctionLink(X86JITCore* code, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
using JCC = void (X86JITCore::*)(const Label& label, LabelType type);
|
||||
@@ -274,9 +308,6 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(GuestReturn);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -285,7 +316,7 @@ private:
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -320,7 +351,7 @@ private:
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -329,6 +360,8 @@ private:
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -338,8 +371,6 @@ private:
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(CreateVector2);
|
||||
DEF_OP(CreateVector4);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
@@ -426,6 +457,7 @@ private:
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
@@ -434,6 +466,8 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
+39
-38
@@ -4,6 +4,7 @@ tags: backend|x86-64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -23,7 +24,7 @@ DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
@@ -84,23 +85,23 @@ DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Header.Args[0].ID()));
|
||||
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Header.Args[0].ID()));
|
||||
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
case 16:
|
||||
@@ -112,27 +113,27 @@ DEF_OP(StoreContext) {
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
pextrw(word [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
|
||||
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
|
||||
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
|
||||
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
|
||||
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
@@ -143,9 +144,9 @@ DEF_OP(StoreContext) {
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
Reg index = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
@@ -245,11 +246,11 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
Reg index = GetSrc<RA_64>(Op->Header.Args[1].ID());
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
size_t size = IROp->Size;
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
auto value = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto value = GetSrc<RA_64>(Op->Value.ID());
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -269,7 +270,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Header.Args[0].ID());
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
@@ -339,19 +340,19 @@ DEF_OP(SpillRegister) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
mov(byte [rsp + SlotOffset], GetSrc<RA_8>(Op->Header.Args[0].ID()));
|
||||
mov(byte [rsp + SlotOffset], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
mov(word [rsp + SlotOffset], GetSrc<RA_16>(Op->Header.Args[0].ID()));
|
||||
mov(word [rsp + SlotOffset], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
mov(dword [rsp + SlotOffset], GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov(dword [rsp + SlotOffset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -359,15 +360,15 @@ DEF_OP(SpillRegister) {
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(dword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
|
||||
movss(dword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(qword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
|
||||
movsd(qword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(xword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
|
||||
movaps(xword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -435,7 +436,7 @@ DEF_OP(LoadFlag) {
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
mov (rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov (rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
}
|
||||
|
||||
@@ -469,7 +470,7 @@ DEF_OP(LoadMem) {
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
@@ -537,19 +538,19 @@ DEF_OP(StoreMem) {
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
case 2:
|
||||
mov(word [MemPtr], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
case 4:
|
||||
mov(dword [MemPtr], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
@@ -557,22 +558,22 @@ DEF_OP(StoreMem) {
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Header.Args[1].ID()), 0);
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(word [MemPtr], GetSrc(Op->Header.Args[1].ID()), 0);
|
||||
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(dword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
case 8:
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
case 16:
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
|
||||
+58
-70
@@ -7,6 +7,7 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -18,16 +19,14 @@ $end_info$
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, getCurr<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
@@ -46,62 +45,30 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
ud2();
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
|
||||
jmp(TMP1);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddress);
|
||||
jmp(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_Interrupt3: // INT3
|
||||
{
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// This jump target needs to be a constant offset here
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
|
||||
jmp(TMP1);
|
||||
}
|
||||
else {
|
||||
// If we don't have a gdb server attached then....crash?
|
||||
// Treat this case like HLT
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)], Op->Reason.Signal);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], Op->Reason.TrapNumber);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], Op->Reason.ErrorRegister);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], Op->Reason.si_code);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddress);
|
||||
jmp(TMP1);
|
||||
}
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)]);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)]);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)]);
|
||||
break;
|
||||
default:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)]);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
|
||||
jmp(TMP1);
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -117,7 +84,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrc<RA_32>(Op->RoundMode.ID());
|
||||
|
||||
// Load old mxcsr
|
||||
// Only stores to memory
|
||||
@@ -142,20 +109,17 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushRegs();
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
mov(rax, reinterpret_cast<uintptr_t>(PrintValue));
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Value.ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)]);
|
||||
}
|
||||
else {
|
||||
pextrq(rdi, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
pextrq(rdi, GetSrc(Op->Value.ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Value.ID()), 1);
|
||||
|
||||
mov(rax, reinterpret_cast<uintptr_t>(PrintVectorValue));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)]);
|
||||
}
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
}
|
||||
|
||||
@@ -166,6 +130,27 @@ DEF_OP(ProcessorID) {
|
||||
mov (GetDst<RA_32>(Node), ecx);
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
rdrand(Dst.first);
|
||||
}
|
||||
else {
|
||||
rdseed(Dst.first);
|
||||
}
|
||||
|
||||
// In the case of RDRAND or RDSEED returning a valid number then CF = 1, else 0
|
||||
mov (Dst.second, 0);
|
||||
setc(Dst.second.cvt8());
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
pause();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -174,6 +159,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -183,6 +169,8 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,13 +20,13 @@ DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<Xbyak::Reg, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetDst<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<Xbyak::Reg, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetDst<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
@@ -42,18 +42,18 @@ DEF_OP(CreateElementPair) {
|
||||
Xbyak::Reg RegSecond;
|
||||
Xbyak::Reg RegTmp;
|
||||
|
||||
switch (Op->Header.Size) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetSrc<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Upper.ID());
|
||||
RegTmp = eax;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetSrc<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Upper.ID());
|
||||
RegTmp = rax;
|
||||
break;
|
||||
}
|
||||
@@ -75,7 +75,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+413
-320
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,139 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|x86-64
|
||||
desc: relocation logic of the x86-64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
uint64_t X86JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
|
||||
}
|
||||
|
||||
void X86JITCore::LoadConstantWithPadding(Xbyak::Reg Reg, uint64_t Constant) {
|
||||
// The maximum size a move constant can be in bytes
|
||||
// Need to NOP pad to this size to ensure backpatching is always the same size
|
||||
// Calculated as:
|
||||
// [Rex]
|
||||
// [Mov op]
|
||||
// [8 byte constant]
|
||||
//
|
||||
// All other move types are smaller than this. xbyak will use a NOP slide which is quite quick
|
||||
constexpr static size_t MAX_MOVE_SIZE = 10;
|
||||
auto StartingOffset = getSize();
|
||||
mov(Reg, Constant);
|
||||
auto MoveSize = getSize() - StartingOffset;
|
||||
auto NOPPadSize = MAX_MOVE_SIZE - MoveSize;
|
||||
nop(NOPPadSize);
|
||||
}
|
||||
|
||||
X86JITCore::NamedSymbolLiteralPair X86JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
NamedSymbolLiteralPair Lit {
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void X86JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = getSize();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CursorEntry;
|
||||
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Lit.MoveABI.NamedSymbolLiteral.Symbol);
|
||||
|
||||
L(Lit.Offset);
|
||||
dq(Pointer);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
|
||||
void X86JITCore::InsertGuestRIPMove(Xbyak::Reg Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = getSize();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CursorEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.getIdx();
|
||||
|
||||
if (CTX->Config.CacheObjectCodeCompilation()) {
|
||||
LoadConstantWithPadding(Reg, Constant);
|
||||
}
|
||||
else {
|
||||
mov(Reg, Constant);
|
||||
}
|
||||
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool X86JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Place the pointer
|
||||
dq(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(CTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstantWithPadding(Xbyak::Reg64(Reloc->NamedThunkMove.RegisterIndex), Pointer);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstantWithPadding(Xbyak::Reg64(Reloc->GuestRIPMove.RegisterIndex), Pointer);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
+5
-10
@@ -37,11 +37,11 @@ LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
LOGMAN_THROW_AA_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
LOGMAN_THROW_AA_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
@@ -52,15 +52,8 @@ LookupCache::~LookupCache() {
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
|
||||
}
|
||||
|
||||
void LookupCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
// Tell the kernel we will definitely need [Address, Address+Size) mapped for the page pointer
|
||||
// Page Pointer is allocated per page, so shift by page size
|
||||
Address >>= 12;
|
||||
Size >>= 12;
|
||||
madvise(reinterpret_cast<void*>(PagePointer + Address), Size, MADV_WILLNEED);
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8, MADV_DONTNEED);
|
||||
madvise(reinterpret_cast<void*>(PageMemory), CODE_SIZE, MADV_DONTNEED);
|
||||
@@ -68,6 +61,8 @@ void LookupCache::ClearL2Cache() {
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1
|
||||
madvise(reinterpret_cast<void*>(L1Pointer), L1_SIZE, MADV_DONTNEED);
|
||||
// Clear L2
|
||||
|
||||
+76
-56
@@ -7,6 +7,7 @@
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -24,38 +25,73 @@ public:
|
||||
LookupCache(FEXCore::Context::Context *CTX);
|
||||
~LookupCache();
|
||||
|
||||
using LookupCacheIter = uintptr_t;
|
||||
uintptr_t End() { return 0; }
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
auto HostCode = FindCodePointerForAddress(Address);
|
||||
if (HostCode) {
|
||||
return HostCode;
|
||||
} else {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
// Try L1, no lock needed
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
} else {
|
||||
return 0;
|
||||
// L2 and L3 need to be locked
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize -1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address)
|
||||
{
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto InsertPoint =
|
||||
#endif
|
||||
BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LOGMAN_THROW_A_FMT(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
CodePages[CurrentPage].push_back(Address);
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto &CodePage = CodePages[CurrentPage];
|
||||
rv |= CodePage.size() == 0;
|
||||
CodePage.push_back(Address);
|
||||
}
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -65,6 +101,8 @@ public:
|
||||
|
||||
void Erase(uint64_t Address) {
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks.lower_bound({Address, 0});
|
||||
auto upper = BlockLinks.upper_bound({Address, UINTPTR_MAX});
|
||||
@@ -78,7 +116,10 @@ public:
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = L1Entry.HostCode = 0;
|
||||
L1Entry.GuestCode = 0;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
// Do full map
|
||||
@@ -101,14 +142,14 @@ public:
|
||||
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks.insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
|
||||
void HintUsedRange(uint64_t Address, uint64_t Size);
|
||||
|
||||
uintptr_t GetL1Pointer() const { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() const { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
@@ -116,8 +157,19 @@ public:
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
@@ -167,38 +219,6 @@ private:
|
||||
return PageMemory + NewBase;
|
||||
}
|
||||
|
||||
uintptr_t FindCodePointerForAddress(uint64_t Address) {
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize -1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == FullAddress)
|
||||
{
|
||||
L1Entry.GuestCode = FullAddress;
|
||||
return L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
}
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
|
||||
+84
@@ -0,0 +1,84 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie{};
|
||||
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock{};
|
||||
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
bool MultiBlock : 1;
|
||||
|
||||
// TSO enabled
|
||||
bool TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
bool ABILocalFlags : 1;
|
||||
|
||||
// ABI no PF unsafe optimization
|
||||
bool ABINoPF : 1;
|
||||
|
||||
// Static register allocation enabled
|
||||
bool SRA : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
bool ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
bool Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
bool x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
SRA == other.SRA &&
|
||||
ParanoidTSO == other.ParanoidTSO &&
|
||||
Is64BitMode == other.Is64BitMode &&
|
||||
SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(CodeObjectSerializationConfig const &other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash{};
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
Hash <<= 1; Hash |= other.SRA;
|
||||
Hash <<= 1; Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1; Hash |= other.Is64BitMode;
|
||||
Hash <<= 2; Hash |= other.SMCChecks;
|
||||
Hash <<= 1; Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <sys/uio.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
auto BaseFilename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = std::make_unique<CodeRegionEntry>(
|
||||
Base,
|
||||
Size,
|
||||
Offset,
|
||||
filename,
|
||||
NamedRegionHandler->DefaultCodeHeader(Base, Offset)
|
||||
);
|
||||
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
std::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
}
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::Context *ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ABINoPF = ctx->Config.ABINoPF;
|
||||
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion *>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion *>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService *This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler { &NamedRegionHandler , this }
|
||||
, NamedRegionHandler { ctx } {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
WorkerThreadShuttingDown = true;
|
||||
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, std::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
CodeObjectFileSection const *CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16] = "ObjectCodeSeri\0";
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto &it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
}
|
||||
Loaded 100 of 708 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user