mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 17:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8c956e6ce1 | ||
|
|
832d013c92 | ||
|
|
1c24206117 | ||
|
|
f1979c15a2 | ||
|
|
556a1dab24 | ||
|
|
caffad8562 | ||
|
|
5978143141 | ||
|
|
594c70b5e0 | ||
|
|
655e6989ca | ||
|
|
afeb228a89 | ||
|
|
d308a438ea | ||
|
|
260fc8ba52 | ||
|
|
ade0d0f241 | ||
|
|
11a5105547 | ||
|
|
10ad5db686 | ||
|
|
68c441575d | ||
|
|
334a8ef87c | ||
|
|
b7a76af72f | ||
|
|
4c92b562b8 | ||
|
|
9d08451903 | ||
|
|
c252f8bfc5 | ||
|
|
dc7ec6377b | ||
|
|
ce6f4edaaa | ||
|
|
983c35ea3b | ||
|
|
70aaa1117a | ||
|
|
59e9859087 | ||
|
|
d9453ff639 | ||
|
|
70754991d1 | ||
|
|
57ebfceb48 | ||
|
|
9e224d2bb0 | ||
|
|
e43bd04901 | ||
|
|
48762e03a6 | ||
|
|
858924309e | ||
|
|
cab02d1e65 | ||
|
|
2a64f80567 | ||
|
|
174ddea99d | ||
|
|
6fb0e3c0cf | ||
|
|
2b044bbdf4 | ||
|
|
ea76de0fd2 | ||
|
|
9c8642e0dc | ||
|
|
13f35f7b79 | ||
|
|
e2798e370e | ||
|
|
d9149548b5 | ||
|
|
4823933f79 | ||
|
|
ee04067424 | ||
|
|
e4aef26ef5 | ||
|
|
a41dc8eafa | ||
|
|
f41cd8deff | ||
|
|
2a0c3cce30 | ||
|
|
e817f5d98c | ||
|
|
8b8cda9b80 | ||
|
|
8e3893df07 | ||
|
|
51b335914c | ||
|
|
8835d57ae3 | ||
|
|
eba1b65fb0 | ||
|
|
a3a138ef7e | ||
|
|
62d9a494cd | ||
|
|
6744a06a53 | ||
|
|
140e9824b7 | ||
|
|
bad84f61fa | ||
|
|
2296126af3 | ||
|
|
0a8717d9a8 | ||
|
|
4e2220c27f | ||
|
|
c87e11cee9 | ||
|
|
7768f6965a | ||
|
|
023aaaae0c | ||
|
|
a2aa9f3fc1 | ||
|
|
e46ec9a0ce | ||
|
|
784cbdd973 | ||
|
|
6022715a9b | ||
|
|
9eb5ba5ad1 | ||
|
|
82e5977709 | ||
|
|
7c08b67dff | ||
|
|
609587f9ee | ||
|
|
5dda3a1599 | ||
|
|
73aaa4c3a6 | ||
|
|
3ba5371d36 | ||
|
|
bf581decde | ||
|
|
250504502a | ||
|
|
a0e826feaf | ||
|
|
eb17edec05 | ||
|
|
228aed98c7 | ||
|
|
6ba4aec88e | ||
|
|
b8d6b2cd4a | ||
|
|
d4e2f42f90 | ||
|
|
3055c23365 | ||
|
|
cb7feaefbb | ||
|
|
5a5a498ed6 | ||
|
|
285ed8f1e0 | ||
|
|
a68da468a5 | ||
|
|
d59aa6874e | ||
|
|
19fd89d2bf | ||
|
|
e36beb8dbe | ||
|
|
70f447b265 | ||
|
|
a0026c92a8 | ||
|
|
8fc5f66a5b | ||
|
|
affbb40cc1 | ||
|
|
b276750c8a | ||
|
|
6d5322b349 | ||
|
|
f836ff0897 | ||
|
|
0bf577654d | ||
|
|
ac63a2bac4 | ||
|
|
6b8e6c18ee | ||
|
|
93dccffc81 | ||
|
|
22810208b4 | ||
|
|
154468c172 | ||
|
|
0e45b98fb2 | ||
|
|
2d573f7bf9 | ||
|
|
01587270d9 | ||
|
|
0a63a15287 | ||
|
|
a65e7a09e7 | ||
|
|
7e5c561849 | ||
|
|
a45bc7598c | ||
|
|
9f8071ce8f | ||
|
|
0808599813 | ||
|
|
340699ca33 | ||
|
|
00a572c00c | ||
|
|
d300181ab9 | ||
|
|
5146651908 | ||
|
|
875bae41a3 | ||
|
|
95bd309db9 | ||
|
|
6f1e7e7a4c | ||
|
|
b078f51199 | ||
|
|
d03856f0b7 | ||
|
|
98210fdf47 | ||
|
|
3d1cef4afe | ||
|
|
060ab98378 | ||
|
|
1d5bd1520a | ||
|
|
9eca823e84 | ||
|
|
cd4269f4e8 | ||
|
|
2f1d44f838 | ||
|
|
95eb456065 | ||
|
|
d4b31cd4c6 | ||
|
|
c611eb228f | ||
|
|
bc120b9f97 | ||
|
|
995672921f | ||
|
|
514b6a822f | ||
|
|
0945c728e9 | ||
|
|
3a3e2776ba | ||
|
|
b72245a2d5 | ||
|
|
9c49bd3c8e | ||
|
|
c691d70919 | ||
|
|
f3a27a57f1 | ||
|
|
edce981824 | ||
|
|
9bffaeea40 | ||
|
|
88ce9b5cd2 | ||
|
|
72e8a997f6 | ||
|
|
d708cbad5e | ||
|
|
152eaff00b | ||
|
|
2079f6b3c7 | ||
|
|
2c31080fd4 | ||
|
|
ef7b77dff7 | ||
|
|
777aadb73e | ||
|
|
ccd06e2097 | ||
|
|
09a5f8c6b5 | ||
|
|
8f835678f5 | ||
|
|
8f2cd39802 | ||
|
|
059e8a9099 | ||
|
|
5fae07f6ee | ||
|
|
02005e71ca | ||
|
|
59cc5228b8 | ||
|
|
117cbde226 | ||
|
|
565d1e27d7 | ||
|
|
22ecff05c9 | ||
|
|
17480c0e2d | ||
|
|
0608d95322 | ||
|
|
47bd47950b | ||
|
|
9aab7e07f4 | ||
|
|
f2fd9d9e3f | ||
|
|
45430e9569 | ||
|
|
be3e3a351a | ||
|
|
2cee9e5d4b | ||
|
|
03cc35f341 | ||
|
|
3aead5ef65 | ||
|
|
b9abd091d5 | ||
|
|
43dc232e5a | ||
|
|
ea73c9d7ea | ||
|
|
fa6f1b1d90 | ||
|
|
e24eb7a72d | ||
|
|
e05d116b05 | ||
|
|
ee821b9cbf | ||
|
|
c29e563836 | ||
|
|
fd59fb1a7a | ||
|
|
04deeb3911 | ||
|
|
4a7c65b20d | ||
|
|
23cb0dea00 | ||
|
|
bcad7a9eea | ||
|
|
0a3a270a63 | ||
|
|
06e4a5a5b7 | ||
|
|
6ff80670b3 | ||
|
|
38eea80b8d | ||
|
|
c43af0e10f | ||
|
|
2fa8cd7e97 | ||
|
|
5d0734a7f2 | ||
|
|
3e0e922fef | ||
|
|
e2004f4999 | ||
|
|
13f37efe12 | ||
|
|
80e66a36eb | ||
|
|
4998d35ec5 | ||
|
|
f0655874e3 | ||
|
|
9394e49c95 | ||
|
|
91984003b6 | ||
|
|
dbf571fdfb | ||
|
|
b6abcc5e3c | ||
|
|
e99a23cfc6 | ||
|
|
425d9323d5 | ||
|
|
a7c0997daf | ||
|
|
6eba3f331e | ||
|
|
bcc75e3312 | ||
|
|
30672e1517 | ||
|
|
b6e46fd44d | ||
|
|
32a0e37569 | ||
|
|
5b6175b702 | ||
|
|
0285e35c87 | ||
|
|
bbfb8713a7 | ||
|
|
c1d6967cb6 | ||
|
|
20e24eae9c | ||
|
|
b8d9027680 | ||
|
|
688ef9f5af | ||
|
|
181c9074ab | ||
|
|
6f85f64bfd | ||
|
|
ba6ee61db2 | ||
|
|
e735281b2c | ||
|
|
c8b2d714a8 | ||
|
|
bba0625a57 | ||
|
|
c9a7b36210 | ||
|
|
99bf02f27f | ||
|
|
a2c02ae51e | ||
|
|
c7b59143c6 | ||
|
|
1e23e61572 | ||
|
|
f4dd1a895e | ||
|
|
3a2f9a4d46 | ||
|
|
4a848d7202 | ||
|
|
1b58ed9f57 | ||
|
|
1c86f7ed36 | ||
|
|
2733b2ee1e | ||
|
|
4ceb2dfdf2 | ||
|
|
bd380e0f15 | ||
|
|
73ec786c60 | ||
|
|
e3a2c8dc80 | ||
|
|
d3b14df840 | ||
|
|
a12ab8f98a | ||
|
|
966b9a69d8 | ||
|
|
927d3d00e2 | ||
|
|
bfb9cabeb8 | ||
|
|
22466a973c | ||
|
|
c05e1c9797 | ||
|
|
c65be9f55d | ||
|
|
edc31fe0a3 | ||
|
|
d4655fbb17 | ||
|
|
5f0dfcd715 | ||
|
|
e62cb417c2 | ||
|
|
df486c0786 | ||
|
|
9d43904792 | ||
|
|
5654f9a030 | ||
|
|
d74cf6d8d8 | ||
|
|
84905d2856 | ||
|
|
92cb9d477e | ||
|
|
7ed6007252 | ||
|
|
b6499ac724 | ||
|
|
c73991f467 | ||
|
|
0bf0fe4779 | ||
|
|
097f48f3ff | ||
|
|
75e5df545a | ||
|
|
171d5f7263 | ||
|
|
970067d19b | ||
|
|
c26ff60949 | ||
|
|
04d830ed39 | ||
|
|
b9c49027c7 | ||
|
|
9860e8b71b | ||
|
|
a1e94a9863 | ||
|
|
2945c13dcb | ||
|
|
4f68821aef | ||
|
|
13087f8425 | ||
|
|
fd0424768b | ||
|
|
5c4112f103 | ||
|
|
e8670bbab2 | ||
|
|
13b14b857b | ||
|
|
48c7ff2a23 | ||
|
|
77a032a286 | ||
|
|
f2de640395 | ||
|
|
9a642158e0 | ||
|
|
dc44caa178 | ||
|
|
081a003c6c | ||
|
|
4c5fc6e813 | ||
|
|
253cdb552d | ||
|
|
6404aba6e2 | ||
|
|
45f919683a | ||
|
|
c3a6890a40 | ||
|
|
07be5a0bae | ||
|
|
d0943322b8 | ||
|
|
d6e4da7e77 | ||
|
|
ed5d7f6a62 | ||
|
|
18b223811a | ||
|
|
e1d21f0bff | ||
|
|
d2783f2edd | ||
|
|
3377e5a50e | ||
|
|
db0740becc | ||
|
|
ec8f077e60 | ||
|
|
2b190f1713 | ||
|
|
69413d51bf | ||
|
|
8c38f936da | ||
|
|
e3fbe48f7c | ||
|
|
c1acd9ad59 | ||
|
|
381d07707c | ||
|
|
a07e77b028 | ||
|
|
6b5ccb8244 | ||
|
|
597e4e1d57 | ||
|
|
cdab71e612 | ||
|
|
8ef0278338 | ||
|
|
98f54eab9c | ||
|
|
919e2bc3bd | ||
|
|
ffee22d3fd | ||
|
|
e648e90c2e | ||
|
|
b93871ff55 | ||
|
|
352f0a1133 | ||
|
|
6c93b6f775 | ||
|
|
f806577c46 | ||
|
|
fac4022302 | ||
|
|
b5b3cb252a | ||
|
|
d5bf6414f3 | ||
|
|
fc858ca0a5 | ||
|
|
8ddf1ccbfe | ||
|
|
f2292c08e9 | ||
|
|
b1b616087b | ||
|
|
5542360948 | ||
|
|
7490d15369 | ||
|
|
68587cd02c | ||
|
|
a9e1d5f2a2 | ||
|
|
a8b9b351cb | ||
|
|
559f7128c3 | ||
|
|
5c1f14a155 | ||
|
|
82ebdf52be | ||
|
|
86684f9033 | ||
|
|
e6d285a716 | ||
|
|
5c98c782f6 | ||
|
|
9b43f94f51 | ||
|
|
19a0651944 | ||
|
|
436d20d660 | ||
|
|
e027b83a06 | ||
|
|
d0aa785e6f | ||
|
|
b2f44d607f | ||
|
|
0aa53f726a | ||
|
|
b7e2dcc6c1 | ||
|
|
90e9010b17 | ||
|
|
c62945e615 | ||
|
|
c78976adf1 | ||
|
|
13ad5b66bb | ||
|
|
25b9bcc652 | ||
|
|
e2e2687a85 | ||
|
|
2d3aca2398 | ||
|
|
a1d1674033 | ||
|
|
a281b6b38b | ||
|
|
9e0c652a9a | ||
|
|
fdc0ce7101 | ||
|
|
600b1ad88a | ||
|
|
b45603d174 | ||
|
|
28cb1240b4 | ||
|
|
63f9e0b410 | ||
|
|
e4c1a285ce | ||
|
|
1f306d666c | ||
|
|
baadf0b98e | ||
|
|
0c697af5fb | ||
|
|
e6ad608226 | ||
|
|
1cf0c335a6 | ||
|
|
52ed4a97cf | ||
|
|
db6cbbe7d4 | ||
|
|
70a66cdff2 | ||
|
|
ef772c3ef4 | ||
|
|
8ab6830fef | ||
|
|
5d62a8cedc | ||
|
|
262c1c55d2 | ||
|
|
6913a8b5ec | ||
|
|
5345f7fa37 | ||
|
|
966272ceaf | ||
|
|
bb881c5c9b | ||
|
|
92ae6ae2e4 | ||
|
|
7bb789121b | ||
|
|
97e3a36427 | ||
|
|
0560e9be4e | ||
|
|
513b02fda3 | ||
|
|
0511764df3 | ||
|
|
38b9e85f4f | ||
|
|
75b2f226f6 | ||
|
|
c4d05fb8e3 | ||
|
|
e0d40dd403 | ||
|
|
f97fdd4593 | ||
|
|
6b6b5e880c | ||
|
|
83dc458019 | ||
|
|
4623e4ca21 | ||
|
|
bc442871b4 | ||
|
|
abaddcccf4 | ||
|
|
ee912c1bb8 | ||
|
|
f69013fd44 | ||
|
|
33be4fa98d | ||
|
|
99f760dcab | ||
|
|
16f1ad4692 | ||
|
|
e90602f79c | ||
|
|
f2d06ecf67 | ||
|
|
23e583e131 | ||
|
|
d883b4bf1c | ||
|
|
f584f16ca4 | ||
|
|
fe3164924c | ||
|
|
826818192d | ||
|
|
46e30495dc | ||
|
|
702c9e97e4 | ||
|
|
2bdd100d2c | ||
|
|
97ba05ca5b | ||
|
|
508d72d6cc | ||
|
|
3183cf79e5 | ||
|
|
5d77a64c8f | ||
|
|
005c3ce4b3 | ||
|
|
41adfebf9d | ||
|
|
e1604fb32f | ||
|
|
3730a4284c | ||
|
|
fbd14b65f7 | ||
|
|
6e1ea92c09 | ||
|
|
e68d4c52f7 | ||
|
|
5759b0d503 | ||
|
|
b10ee67525 | ||
|
|
47d04ae807 | ||
|
|
3dc938b8c5 | ||
|
|
d81d5e21ba | ||
|
|
a92233c7f0 | ||
|
|
61ccec3726 | ||
|
|
08bfc5e0c0 | ||
|
|
f62ec61e4f | ||
|
|
06881af363 | ||
|
|
598533555d | ||
|
|
85651ad090 | ||
|
|
25545dcd66 | ||
|
|
a5ac66c1ce | ||
|
|
b6effa7a9b | ||
|
|
730ba42cf3 | ||
|
|
0165329285 | ||
|
|
f12b4aa3d1 | ||
|
|
a9852d31e7 | ||
|
|
265e8b4d39 | ||
|
|
ef3338ec0d | ||
|
|
d47182b631 | ||
|
|
9ac9b89ab3 | ||
|
|
50c8d9edab | ||
|
|
77c969b424 | ||
|
|
95aabc0947 | ||
|
|
a04cc4dc96 | ||
|
|
2a894bc111 | ||
|
|
f746870356 | ||
|
|
70f8779793 | ||
|
|
1615ed9ed7 | ||
|
|
e3a7c14740 | ||
|
|
66a5057914 | ||
|
|
62dfccc989 | ||
|
|
51d8bb9020 | ||
|
|
65e28ba6cf | ||
|
|
c181033263 | ||
|
|
f371b04d7c | ||
|
|
98f42aafa9 | ||
|
|
69437eaf11 | ||
|
|
c63c5fb664 | ||
|
|
ec8e24e84a | ||
|
|
c6b26739db | ||
|
|
f9e22432d2 | ||
|
|
d54b9cd272 | ||
|
|
92b29138ba | ||
|
|
bab4f53b59 | ||
|
|
aae1dd4d81 | ||
|
|
977d92b7bc | ||
|
|
e1cf60c2ab | ||
|
|
cabb58ad29 | ||
|
|
44f27c81d3 | ||
|
|
c9c8c38d86 | ||
|
|
5c8c36c5d2 | ||
|
|
d413cf8d62 | ||
|
|
18623d07aa | ||
|
|
e63c832a5a | ||
|
|
42161b20a4 | ||
|
|
081d0169d8 | ||
|
|
77093a7f4d | ||
|
|
bb7f542813 | ||
|
|
2f28548ce4 | ||
|
|
498a845c0b | ||
|
|
59ed91c50a | ||
|
|
ad864e0e52 | ||
|
|
9849aef5bc | ||
|
|
11c3ae19b4 | ||
|
|
4351ec9f4e | ||
|
|
59469705e4 | ||
|
|
763e7c5e08 | ||
|
|
2edb5d3004 | ||
|
|
c6662a46b5 | ||
|
|
d1fa8bcfba | ||
|
|
3a3cbf2b5f | ||
|
|
a0eb2eb2e2 | ||
|
|
d58952d1ef | ||
|
|
05ce65322f | ||
|
|
3e5037042f | ||
|
|
6d82410f44 | ||
|
|
49966f2954 | ||
|
|
51d21d861d | ||
|
|
6a79b5cf56 | ||
|
|
1d88b499e0 | ||
|
|
b327daf5f5 | ||
|
|
90511ddd19 | ||
|
|
4be8e23299 | ||
|
|
d4a7afbf5c | ||
|
|
d3cd354c48 | ||
|
|
5d707cf831 | ||
|
|
e37a0bbd34 | ||
|
|
9df8245ef2 | ||
|
|
597d524f9e | ||
|
|
b4a71a2144 | ||
|
|
09ee6d3bf4 | ||
|
|
7d2b3d0846 | ||
|
|
e0973e19fc | ||
|
|
ff9190204c | ||
|
|
decd8bec31 | ||
|
|
a393d6609f | ||
|
|
8aebbbd0ca | ||
|
|
8b64546579 | ||
|
|
95457bc78c | ||
|
|
a9d31227bf | ||
|
|
cae2f8cac4 | ||
|
|
d7764d37db | ||
|
|
bf5042bdf0 | ||
|
|
61a0508ff6 | ||
|
|
4ebbca45be | ||
|
|
8244d55276 | ||
|
|
4b47e66135 | ||
|
|
df2f1ad074 | ||
|
|
8f170d4aa0 | ||
|
|
285ef38717 | ||
|
|
be12059e8f | ||
|
|
b47cb20619 | ||
|
|
166c96320c | ||
|
|
ff24fe872d | ||
|
|
e317424b86 | ||
|
|
babb81a240 | ||
|
|
235367b67a | ||
|
|
56e5e78b25 | ||
|
|
9e8af23456 | ||
|
|
5b9da4f2be | ||
|
|
1f4a10ef1f | ||
|
|
4c712ca111 | ||
|
|
3bcc8ca695 | ||
|
|
83073a880a | ||
|
|
34b2f93ddf | ||
|
|
49dae08b3d | ||
|
|
0b700de7d9 | ||
|
|
43454abc63 | ||
|
|
76538be0e0 | ||
|
|
aa1c47cd75 | ||
|
|
002867bc2a | ||
|
|
79d6bf2840 | ||
|
|
31030e6f85 | ||
|
|
26d493a66e | ||
|
|
e0343647c9 | ||
|
|
33151e16a2 | ||
|
|
7e9201cf0d | ||
|
|
877db85428 | ||
|
|
f9078f8ded | ||
|
|
e547f0cad6 | ||
|
|
a3b39afef2 | ||
|
|
43431edd45 | ||
|
|
c59efaef7a | ||
|
|
0bfc1bbe70 | ||
|
|
e9937d9a85 | ||
|
|
f088f0a236 | ||
|
|
a40a0cbb12 | ||
|
|
28d084bf78 | ||
|
|
435137e1a2 | ||
|
|
ff74e0a0ad | ||
|
|
d847f6e1b3 | ||
|
|
64aa4f00ca | ||
|
|
50c165d291 | ||
|
|
eb8a8bf929 | ||
|
|
17fd5f7f79 | ||
|
|
c9c352627f | ||
|
|
e670f8f0e6 | ||
|
|
c431cdebcc | ||
|
|
8b3c46154d | ||
|
|
1d9b66044a | ||
|
|
031fa8a7d6 | ||
|
|
c9621da51c | ||
|
|
b1ab252c68 | ||
|
|
d09706aa1c | ||
|
|
eb8ca16402 | ||
|
|
8df16460d1 | ||
|
|
5758c65983 | ||
|
|
fa1648c6d5 | ||
|
|
98ba0bfa82 | ||
|
|
366122338e | ||
|
|
bdc66a33ef | ||
|
|
48955da5f3 | ||
|
|
6cd73a6724 | ||
|
|
8dfe305aab | ||
|
|
6fb0b3d85c | ||
|
|
69b27d7715 | ||
|
|
cffd10d0f7 | ||
|
|
f2ef58630c | ||
|
|
de8d8d8751 | ||
|
|
27072d2853 | ||
|
|
0dc8e23342 | ||
|
|
b102714d5c | ||
|
|
72125c9ccf | ||
|
|
9a07b550f3 | ||
|
|
351412a3e3 | ||
|
|
842ab169ce | ||
|
|
b3efb1d2d7 | ||
|
|
0d0ce38050 | ||
|
|
cb21f52f93 | ||
|
|
14b0cc5af4 | ||
|
|
c0fc4c4623 | ||
|
|
c3230f6a91 | ||
|
|
4a3bd7cd13 | ||
|
|
f98627da32 | ||
|
|
f3c20f2743 | ||
|
|
ec6140fde0 | ||
|
|
c5490821fc | ||
|
|
c0aad64578 | ||
|
|
7381240fb6 | ||
|
|
3ea8e4864d | ||
|
|
7bf8d09391 | ||
|
|
798c6772b7 | ||
|
|
29debcda3d | ||
|
|
39c1751215 | ||
|
|
9451bc5273 | ||
|
|
e2b24f7f59 | ||
|
|
4589876ebc | ||
|
|
e0b878f1fd | ||
|
|
073224ffea | ||
|
|
00511c16a4 | ||
|
|
66c7fdb6de | ||
|
|
a48ed376d8 | ||
|
|
98bee5a6dd | ||
|
|
8d13261ab6 | ||
|
|
2b757a9b9a | ||
|
|
08f56540d2 | ||
|
|
fb01e8bf28 | ||
|
|
d5f9f43ab0 | ||
|
|
9d460e807c | ||
|
|
185f265f5c | ||
|
|
591fc001cc | ||
|
|
d71add78ae | ||
|
|
d93edb2da8 | ||
|
|
2360f9cec1 | ||
|
|
10d596314e | ||
|
|
cab0cf6a6b | ||
|
|
6162a8c7f4 | ||
|
|
1b0d2bbf9f | ||
|
|
121023fb72 | ||
|
|
c30cb87b01 | ||
|
|
d8edbba71c | ||
|
|
7c553f3508 | ||
|
|
5b2d944886 | ||
|
|
9ab7de56ef | ||
|
|
fc46cb9390 | ||
|
|
8cf4b285bf | ||
|
|
3744ec2a44 | ||
|
|
486c62f77c | ||
|
|
327c4d550a | ||
|
|
f14b73689f | ||
|
|
93de38d13c | ||
|
|
3107898f06 | ||
|
|
21ff433999 | ||
|
|
217e4764c4 | ||
|
|
f232dcebfc | ||
|
|
76dd09ea60 | ||
|
|
368095f96f | ||
|
|
2cd844bbcb | ||
|
|
55a9ee702b | ||
|
|
cb57797550 | ||
|
|
1b6cbd39a4 | ||
|
|
d004fee7e4 | ||
|
|
1fa3afd0c1 | ||
|
|
ba174e5c32 | ||
|
|
e91061f7ff | ||
|
|
a56463a7f1 | ||
|
|
56bddd3a22 | ||
|
|
829db6c30d | ||
|
|
7855b58c73 | ||
|
|
2b91108255 | ||
|
|
593be950de | ||
|
|
360c4a2060 | ||
|
|
a5046e92cd | ||
|
|
793b25f93b | ||
|
|
acfcfa127a | ||
|
|
35d09de8c6 | ||
|
|
1f11e307ec | ||
|
|
c9a62658e3 | ||
|
|
4ccd68af8a | ||
|
|
cd4586b67d | ||
|
|
2c02dcac9f | ||
|
|
dec512187e | ||
|
|
dd34316562 | ||
|
|
5f7532c569 | ||
|
|
0fa7af15b2 | ||
|
|
8654d19f02 | ||
|
|
befd0aa9cf | ||
|
|
96cef80a25 | ||
|
|
fd3a88389b | ||
|
|
d8350353d6 | ||
|
|
083d3a464a | ||
|
|
70931bf388 | ||
|
|
4f93259332 | ||
|
|
ad34cddbf2 | ||
|
|
c74620f083 | ||
|
|
eefcde369a | ||
|
|
005818177c | ||
|
|
2ae47eae48 | ||
|
|
1c4503e26a | ||
|
|
425ee98f81 | ||
|
|
9945542375 | ||
|
|
0468bb4496 | ||
|
|
e55e3a58b1 | ||
|
|
25e9585564 | ||
|
|
9ab294b2f8 | ||
|
|
d7f4fe7564 | ||
|
|
e09219e5e5 | ||
|
|
c645d8683a | ||
|
|
eb9d3b11f2 | ||
|
|
7795078f7e | ||
|
|
0b564652d2 | ||
|
|
4f66d3e9dc | ||
|
|
bd7822bbe9 | ||
|
|
0c484ac49c | ||
|
|
5d21a1e6d6 | ||
|
|
d600b34b8e | ||
|
|
4003ede7ee | ||
|
|
127d7c1a9e | ||
|
|
f45de45568 | ||
|
|
5fdb66249b | ||
|
|
bb525e2291 | ||
|
|
de29c65585 | ||
|
|
ef0f2246ac | ||
|
|
76877e08bc | ||
|
|
d2636f63f2 | ||
|
|
c241c2f5e7 | ||
|
|
99e24c284c | ||
|
|
bb3bd3faa1 | ||
|
|
6b3a5470c4 | ||
|
|
d241c925c0 | ||
|
|
378dfcf164 | ||
|
|
9fd558e173 | ||
|
|
3fbc3c347a | ||
|
|
fb0b03808f | ||
|
|
a1aeb0d7ee | ||
|
|
4b6b7495df | ||
|
|
29e30773ee | ||
|
|
3e05544e78 | ||
|
|
61260c2a62 | ||
|
|
109c42a629 | ||
|
|
9140ba28b7 | ||
|
|
1c3be542f7 | ||
|
|
8a331202ee | ||
|
|
f72ecacd86 | ||
|
|
057c1de69d | ||
|
|
10793e89a9 | ||
|
|
f161e3bfb0 | ||
|
|
c206942b59 | ||
|
|
10922293c9 | ||
|
|
38ce5876f8 | ||
|
|
de64db5852 | ||
|
|
097b3ad881 | ||
|
|
c25ae9b2c3 | ||
|
|
b625437071 | ||
|
|
b0c8710b2c | ||
|
|
dfce1dc476 | ||
|
|
77db25fc24 | ||
|
|
fa224a3557 | ||
|
|
b373d0fcfe | ||
|
|
a3490aad5b | ||
|
|
304db72d6b | ||
|
|
b15e0c5f6c | ||
|
|
97f413cfec | ||
|
|
6da3330646 | ||
|
|
5debdf8d57 | ||
|
|
6483740553 | ||
|
|
f131f07612 | ||
|
|
f268a28caf | ||
|
|
6afc3ca13a | ||
|
|
35c664295d | ||
|
|
b7af5c641c | ||
|
|
6b87839437 | ||
|
|
9563b5aa3a | ||
|
|
28b3bc3508 | ||
|
|
8099dfc830 | ||
|
|
926ddabbd1 | ||
|
|
af3af9d048 | ||
|
|
3b8f24d74f | ||
|
|
8931ddc382 | ||
|
|
dd8225be8a | ||
|
|
02c10b9671 | ||
|
|
4d7455989c | ||
|
|
23fb4baf46 | ||
|
|
6e5fc5cdde | ||
|
|
6a08587d2e | ||
|
|
7bfa1c4838 | ||
|
|
84f42f6155 | ||
|
|
e4a230c3ec | ||
|
|
09296fe73e | ||
|
|
d2130e1df3 | ||
|
|
01c49dbb9c | ||
|
|
6d60689ad4 | ||
|
|
63af80fce3 | ||
|
|
43052a5707 | ||
|
|
db5a26991f | ||
|
|
c07b5e480b | ||
|
|
7aae9b7e44 | ||
|
|
e90892a20b | ||
|
|
a46773a9ff | ||
|
|
740270c05f | ||
|
|
83d20d8f34 | ||
|
|
e388729403 | ||
|
|
e3fda9f232 | ||
|
|
f5940df822 | ||
|
|
e6c4f9aad1 | ||
|
|
8eb0df96f9 | ||
|
|
55d981fcb0 | ||
|
|
b120a8ea84 | ||
|
|
5d73ac3234 | ||
|
|
b05adaeba3 | ||
|
|
0be16baebf | ||
|
|
8dd41e7fcd | ||
|
|
83cdf0f377 | ||
|
|
b010ab42c7 | ||
|
|
1c1f40e5af | ||
|
|
c6c94570b4 | ||
|
|
cce3f365cc | ||
|
|
4641e44276 |
No files matched your search
@@ -0,0 +1,45 @@
|
||||
---
|
||||
name: Potential Game Bug
|
||||
about: A bug in FEX-Emu that causes a problem in a game
|
||||
title: "[Game]: [Short Problem Description]"
|
||||
labels: Game related
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**What Game**
|
||||
The game name.
|
||||
A link to the storefront where to get the game. GOG, Steam, Itch.io, etc
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behavior:
|
||||
1. Go to '...'
|
||||
2. Click on '....'
|
||||
3. Scroll down to '....'
|
||||
4. See error
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots and Video**
|
||||
If applicable, add screenshots and video to help explain your problem.
|
||||
|
||||
**System information:**
|
||||
- OS: [eg: Ubuntu 21.10]
|
||||
- CPU/SoC: [eg: Snapdragon 888, Intel Core i8-12900k]
|
||||
- Video driver version: [eg: OpenGL ES 3.2 Mesa 22.0.0-devel (git-9ff086052a)]
|
||||
- RootFS used: [eg: Ubuntu 21.10 Official Rootfs]
|
||||
- FEX version: (FEXGetConfig --version) [eg: FEX-2112-155-gc691d709]
|
||||
- Thunks Enabled: [Yes/No]
|
||||
|
||||
**Additional context**
|
||||
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
|
||||
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
|
||||
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
|
||||
- Is this a Vulkan game: [Yes/No/Unknown]
|
||||
- If Yes, What is your Vulkan driver:
|
||||
|
||||
Add any other context about the problem here.
|
||||
@@ -31,7 +31,9 @@ jobs:
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: git submodule update --init --depth 1
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
@@ -49,7 +51,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -140,6 +142,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
|
||||
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
+4
-1
@@ -1,7 +1,7 @@
|
||||
[submodule "External/vixl"]
|
||||
shallow = true
|
||||
path = External/vixl
|
||||
url = https://github.com/Sonicadvance1/vixl.git
|
||||
url = https://github.com/FEX-Emu/vixl.git
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = External/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
@@ -42,3 +42,6 @@
|
||||
[submodule "External/xxhash"]
|
||||
path = External/xxhash
|
||||
url = https://github.com/FEX-Emu/xxHash.git
|
||||
[submodule "External/Catch2"]
|
||||
path = External/Catch2
|
||||
url = https://github.com/catchorg/Catch2.git
|
||||
+158
-69
@@ -16,11 +16,20 @@ option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_STATIC_PIE "Enables static-pie build" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
|
||||
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
|
||||
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
|
||||
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
|
||||
set(ENABLE_ASSERTIONS TRUE)
|
||||
@@ -31,6 +40,11 @@ if (ENABLE_ASSERTIONS)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
message(STATUS "Interpreter enabled")
|
||||
add_definitions(-DINTERPRETER_ENABLED=1)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
|
||||
@@ -76,12 +90,28 @@ if (ENABLE_XRAY)
|
||||
link_libraries(-fxray-instrument)
|
||||
endif()
|
||||
|
||||
if (ENABLE_COMPILE_TIME_TRACE)
|
||||
add_compile_options(-ftime-trace)
|
||||
link_libraries(-ftime-trace)
|
||||
endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
if (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
link_libraries(${LD_OVERRIDE})
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
message(WARNING "This is an unsupported configuration and should only be used for testing")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -stdlib=libc++")
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -stdlib=libc++ -lc++abi")
|
||||
endif()
|
||||
|
||||
if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
# Disable FEX offline telemetry entirely if asked
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_STATIC_PIE)
|
||||
if (_M_ARM_64 AND ENABLE_LLD)
|
||||
message (FATAL_ERROR "Static linking does not currently work with AArch64+LLD. Use GNU ld for now.")
|
||||
@@ -202,7 +232,7 @@ if (ENABLE_STATIC_PIE)
|
||||
message (FATAL_ERROR "Application has __rela_iplt_{start,end} symbols. Which means static-pie can't be enabled")
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Couldn't compile static-pie test. Static-pie can't be enabled!")
|
||||
message (FATAL_ERROR "Couldn't compile static-pie test. Static-pie can't be enabled! Is your glibc compiled without static-pie?")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -219,6 +249,8 @@ endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
else()
|
||||
message (STATUS
|
||||
" jemalloc disabled!\n"
|
||||
@@ -243,19 +275,22 @@ endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
pkg_check_modules(XXHASH libxxhash>=0.8.0 QUIET)
|
||||
|
||||
if (NOT XXHASH_FOUND)
|
||||
message(STATUS "xxHash not found. Using Externals")
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
endif()
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
if (BUILD_TESTS)
|
||||
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
|
||||
set(CATCH_BUILD_STATIC_LIBRARY ON)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/cpp-optparse/)
|
||||
include_directories(External/cpp-optparse/)
|
||||
@@ -295,11 +330,6 @@ if(ENUM_ENUM_WARNING)
|
||||
add_compile_options(-Wno-deprecated-enum-enum-conversion)
|
||||
endif()
|
||||
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
|
||||
endif()
|
||||
|
||||
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
add_compile_options(-Werror)
|
||||
if (NOT ENABLE_STRICT_WERROR)
|
||||
@@ -308,29 +338,52 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=native")
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
add_compile_options("-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${TUNE_CPU}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_IWYU)
|
||||
@@ -403,6 +456,10 @@ if (BUILD_TESTS)
|
||||
enable_testing()
|
||||
message(STATUS "Unit tests are enabled")
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
include_directories(FEXHeaderUtils/)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
@@ -411,20 +468,31 @@ add_subdirectory(Data/binfmts/)
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
# Install the ThunksDB file
|
||||
install(
|
||||
FILES ${CMAKE_CURRENT_SOURCE_DIR}/Data/ThunksDB.json
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
include(ExternalProject)
|
||||
|
||||
ExternalProject_Add(host-libs
|
||||
PREFIX host-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
|
||||
BINARY_DIR "Host"
|
||||
CMAKE_ARGS "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -440,9 +508,16 @@ if (BUILD_THUNKS)
|
||||
PREFIX guest-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}" "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DX86_C_COMPILER:STRING=${X86_C_COMPILER}"
|
||||
"-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -459,59 +534,73 @@ set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
# Parse the version here
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_CONTACT "team@fex-emu.org")
|
||||
if (ENABLE_STATIC_PIE)
|
||||
set (CPACK_PACKAGE_NAME fex-emu-static)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu")
|
||||
else()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu-static")
|
||||
endif()
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.org>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/CPack/Description.txt")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libstdc++6")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA "${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm")
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/CPack/triggers")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "qemu-user-static")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
|
||||
endif()
|
||||
include (CPack)
|
||||
@@ -0,0 +1,3 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX is very much work in progress, so expect things to change.
|
||||
@@ -0,0 +1 @@
|
||||
activate-noawait ldconfig
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"StallProcess": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,349 @@
|
||||
{
|
||||
"DB": {
|
||||
"GL": {
|
||||
"Library" : "libGL-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
"Library": "libGLESv2-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan-radeon": {
|
||||
"Library": "libvulkan_radeon-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_radeon.so"
|
||||
],
|
||||
"Comment": [
|
||||
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
|
||||
]
|
||||
},
|
||||
"Vulkan-lavapipe": {
|
||||
"Library": "libvulkan_lvp-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_lvp.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-freedreno": {
|
||||
"Library": "libvulkan_freedreno-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_freedreno.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-intel": {
|
||||
"Library": "libvulkan_intel-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_intel.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-panfrost": {
|
||||
"Library": "libvulkan_panfrost-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_panfrost.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-nvidia": {
|
||||
"Library": "libvulkan_nvidia-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/lib/x86_64-linux-gnu/libGLX_nvidia.so.0"
|
||||
],
|
||||
"Comment": [
|
||||
"Not currently wired up"
|
||||
]
|
||||
},
|
||||
"Vulkan-virtio": {
|
||||
"Library": "libvulkan_virtio-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_virtio.so"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb_dri2-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb_dri3-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb_xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb_shm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb_sync-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb_randr-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb_present-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb_glx-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
+1
Submodule External/Catch2 added at c4e3767e26.
Vendored
+23
-19
@@ -16,7 +16,6 @@ endif()
|
||||
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
|
||||
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_JITSYMBOLS "Enable visibility of JITSymbols in profiling tools" FALSE)
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
|
||||
@@ -38,27 +37,32 @@ endif()
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
set(GIT_SHORT_HASH "Unknown")
|
||||
set(GIT_DESCRIBE_STRING "FEX-Unknown")
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
endif()
|
||||
else()
|
||||
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
configure_file(
|
||||
|
||||
+20
-1
@@ -374,7 +374,7 @@ def print_parse_argloader_options(options):
|
||||
conversion_func = "std::to_string"
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEX::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = ""
|
||||
@@ -396,6 +396,21 @@ def print_parse_argloader_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
|
||||
def print_parse_envloader_options(options):
|
||||
output_argloader.write("#ifdef ENVLOADER\n")
|
||||
output_argloader.write("#undef ENVLOADER\n")
|
||||
output_argloader.write("if (false) {}\n")
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = {0}(Value);\n".format(conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
@@ -470,4 +485,8 @@ output_man.close()
|
||||
output_argloader = open(output_argumentloader_filename, "w")
|
||||
print_argloader_options(options);
|
||||
print_parse_argloader_options(options);
|
||||
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
output_argloader.close()
|
||||
+28
-14
@@ -7,7 +7,7 @@ def print_enums(ops, defines):
|
||||
output_file.write("enum IROps : uint8_t {\n")
|
||||
|
||||
for op_key, op_vals in ops.items():
|
||||
output_file.write("\t\tOP_%s,\n" % op_key.upper())
|
||||
output_file.write("\tOP_%s,\n" % op_key.upper())
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
@@ -20,7 +20,10 @@ def print_ir_structs(ops, defines):
|
||||
|
||||
# Print out defines here
|
||||
for op_val in defines:
|
||||
output_file.write("\t%s;\n" % op_val)
|
||||
if op_val:
|
||||
output_file.write("\t%s;\n" % op_val)
|
||||
else:
|
||||
output_file.write("\n")
|
||||
|
||||
output_file.write("// Default structs\n")
|
||||
output_file.write("struct __attribute__((packed)) IROp_Header {\n")
|
||||
@@ -81,11 +84,21 @@ def print_ir_structs(ops, defines):
|
||||
|
||||
output_file.write("\tstatic constexpr IROps OPCODE = OP_%s;\n" % op_key.upper())
|
||||
|
||||
if (SSAArgs > 0):
|
||||
# Add helpers for accessing SSA arguments, given how frequently they're accessed
|
||||
output_file.write("\n")
|
||||
output_file.write("\t[[nodiscard]] OrderedNodeWrapper& Args(size_t Index) {\n")
|
||||
output_file.write("\t\treturn Header.Args[Index];\n")
|
||||
output_file.write("\t}\n")
|
||||
output_file.write("\t[[nodiscard]] const OrderedNodeWrapper& Args(size_t Index) const {\n")
|
||||
output_file.write("\t\treturn Header.Args[Index];\n")
|
||||
output_file.write("\t}\n")
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
# Add a static assert that the IR ops must be pod
|
||||
output_file.write("static_assert(std::is_trivial<IROp_%s>::value);\n\n" % op_key)
|
||||
output_file.write("static_assert(std::is_standard_layout<IROp_%s>::value);\n\n" % op_key)
|
||||
output_file.write("static_assert(std::is_trivial_v<IROp_%s>);\n" % op_key)
|
||||
output_file.write("static_assert(std::is_standard_layout_v<IROp_%s>);\n\n" % op_key)
|
||||
|
||||
output_file.write("#undef IROP_STRUCTS\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -106,12 +119,12 @@ def print_ir_sizes(ops, defines):
|
||||
output_file.write("// Make sure our array maps directly to the IROps enum\n")
|
||||
output_file.write("static_assert(IRSizes[IROps::OP_LAST] == -1ULL);\n\n")
|
||||
|
||||
output_file.write("[[maybe_unused]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
|
||||
output_file.write("[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
|
||||
|
||||
output_file.write("__attribute__((const)) __attribute__((visibility(\"default\"))) std::string_view const& GetName(IROps Op);\n")
|
||||
output_file.write("__attribute__((const)) __attribute__((visibility(\"default\"))) uint8_t GetArgs(IROps Op);\n")
|
||||
output_file.write("__attribute__((const)) __attribute__((visibility(\"default\"))) FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("__attribute__((const)) __attribute__((visibility(\"default\"))) bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] std::string_view const& GetName(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -270,7 +283,8 @@ def print_ir_allocator_helpers(ops, defines):
|
||||
output_file.write("\t\t\n")
|
||||
output_file.write("\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n")
|
||||
output_file.write("\t\toperator OrderedNode *() { return Node; }\n")
|
||||
output_file.write("\t\toperator OpNodeWrapper () { return Node->Header.Value; }\n")
|
||||
output_file.write("\t\toperator const OrderedNode *() const { return Node; }\n")
|
||||
output_file.write("\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n")
|
||||
output_file.write("\t};\n")
|
||||
|
||||
output_file.write("\ttemplate <class T>\n")
|
||||
@@ -301,18 +315,18 @@ def print_ir_allocator_helpers(ops, defines):
|
||||
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpSize(OrderedNode *Op) const {\n")
|
||||
output_file.write("\tuint8_t GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElements(OrderedNode *Op) const {\n")
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A(HeaderOp->HasDest, \"Op %s has no dest\\n\", GetName(HeaderOp->Op));\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(HeaderOp->HasDest, \"Op {} has no dest\\n\", GetName(HeaderOp->Op));\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size / HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tbool OpHasDest(OrderedNode *Op) const {\n")
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->HasDest;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
+78
-33
@@ -1,9 +1,15 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (SRCS
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Common/Paths.cpp
|
||||
Interface/Config/Config.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
)
|
||||
|
||||
set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Common/NetStream.cpp
|
||||
Common/SoftFloat-3e/extF80_add.c
|
||||
Common/SoftFloat-3e/extF80_div.c
|
||||
Common/SoftFloat-3e/extF80_sub.c
|
||||
@@ -71,7 +77,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/f32_to_extF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Config/Config.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
@@ -86,6 +91,7 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/Vector.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
@@ -94,8 +100,7 @@ set (SRCS
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/X86Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/InterpreterFallbacks.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -109,6 +114,7 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
@@ -118,6 +124,7 @@ set (SRCS
|
||||
Interface/IR/Passes/DeadContextStoreElimination.cpp
|
||||
Interface/IR/Passes/IRCompaction.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/LongDivideRemovalPass.cpp
|
||||
Interface/IR/Passes/ValueDominanceValidation.cpp
|
||||
Interface/IR/Passes/PhiValidation.cpp
|
||||
@@ -128,10 +135,28 @@ set (SRCS
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/Allocator/64BitAllocator.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/ALUOps.cpp
|
||||
Interface/Core/Interpreter/AtomicOps.cpp
|
||||
Interface/Core/Interpreter/BranchOps.cpp
|
||||
Interface/Core/Interpreter/ConversionOps.cpp
|
||||
Interface/Core/Interpreter/EncryptionOps.cpp
|
||||
Interface/Core/Interpreter/F80Ops.cpp
|
||||
Interface/Core/Interpreter/FlagOps.cpp
|
||||
Interface/Core/Interpreter/MemoryOps.cpp
|
||||
Interface/Core/Interpreter/MiscOps.cpp
|
||||
Interface/Core/Interpreter/MoveOps.cpp
|
||||
Interface/Core/Interpreter/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp)
|
||||
@@ -179,10 +204,16 @@ if (ENABLE_JIT_ARM64)
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JITSYMBOLS)
|
||||
list(APPEND DEFINES -DENABLE_JITSYMBOLS=1)
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
|
||||
|
||||
# Generate IR include file
|
||||
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
@@ -225,8 +256,9 @@ add_custom_target(IR_INC
|
||||
set(OUTPUT_CONFIG_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/Config")
|
||||
set(OUTPUT_CONFIG_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigValues.inl")
|
||||
set(OUTPUT_CONFIG_OPTION_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigOptions.inl")
|
||||
set(INPUT_CONFIG_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json")
|
||||
set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
|
||||
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
|
||||
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
|
||||
|
||||
add_custom_target(CREATE_CONFIG_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
|
||||
@@ -242,6 +274,12 @@ add_custom_command(
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
)
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
|
||||
@@ -249,32 +287,28 @@ set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
|
||||
|
||||
set_source_files_properties(${OUTPUT_MAN_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
set_source_files_properties(${OUTPUT_MAN_NAME_COMPRESS} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
|
||||
# Create the target
|
||||
add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_CONFIG_NAME}"
|
||||
DEPENDS "${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME}")
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
# Install the man page
|
||||
install(FILES ${OUTPUT_MAN_NAME} DESTINATION ${MAN_DIR}/man1)
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
check_cxx_compiler_flag(-fdiagnostics-color=always GCC_COLOR)
|
||||
check_cxx_compiler_flag(-fcolor-diagnostics CLANG_COLOR)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_link_libraries(${Name} vixl dl fmt::fmt xxhash)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
function(AddDefaultOptionsToTarget Name)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
|
||||
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
|
||||
target_include_directories(${Name} PRIVATE IncludePrivate/)
|
||||
@@ -284,6 +318,7 @@ function(AddObject Name Type)
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
target_compile_definitions(${Name} PRIVATE ${DEFINES})
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
@@ -306,19 +341,6 @@ function(AddObject Name Type)
|
||||
PRIVATE
|
||||
"-fcolor-diagnostics")
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} vixl dl fmt::fmt xxhash)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
|
||||
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
target_include_directories(${Name} PUBLIC "${PROJECT_SOURCE_DIR}/include/")
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${Name}
|
||||
@@ -330,6 +352,29 @@ function(AddLibrary Name Type)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base fmt::fmt tiny-json)
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
endfunction()
|
||||
|
||||
AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
+14
-10
@@ -1,13 +1,16 @@
|
||||
#pragma once
|
||||
#include "Common/MathUtils.h"
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
template<typename T>
|
||||
struct BitSet final {
|
||||
using ElementType = T;
|
||||
@@ -17,12 +20,12 @@ struct BitSet final {
|
||||
ElementType *Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
@@ -61,8 +64,8 @@ struct BitSetView final {
|
||||
ElementType *Memory;
|
||||
|
||||
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_A((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
@@ -87,11 +90,12 @@ struct BitSetView final {
|
||||
bool operator[](T Element) {
|
||||
return Get(Element);
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
static_assert(sizeof(BitSet<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
|
||||
static_assert(std::is_trivially_copyable<BitSet<uint32_t>>::value, "Needs to trivially copyable");
|
||||
static_assert(std::is_trivially_copyable_v<BitSet<uint32_t>>, "Needs to trivially copyable");
|
||||
|
||||
static_assert(sizeof(BitSetView<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
|
||||
static_assert(std::is_trivially_copyable<BitSetView<uint32_t>>::value, "Needs to trivially copyable");
|
||||
static_assert(std::is_trivially_copyable_v<BitSetView<uint32_t>>, "Needs to trivially copyable");
|
||||
|
||||
} // namespace FEXCore
|
||||
+29
-21
@@ -1,45 +1,53 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
|
||||
#include <string>
|
||||
#include <sstream>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() {
|
||||
std::stringstream PerfMap;
|
||||
PerfMap << "/tmp/perf-" << getpid() << ".map";
|
||||
#include <fmt/format.h>
|
||||
|
||||
fp = fopen(PerfMap.str().c_str(), "wb");
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fp.reset(fopen(PerfMap.c_str(), "wb"));
|
||||
if (fp) {
|
||||
// Disable buffering on this file
|
||||
setvbuf(fp, nullptr, _IONBF, 0);
|
||||
setvbuf(fp.get(), nullptr, _IONBF, 0);
|
||||
}
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() {
|
||||
if (fp) {
|
||||
fclose(fp);
|
||||
}
|
||||
}
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
std::stringstream String;
|
||||
String << std::hex << HostAddr << " " << CodeSize << " " << "JIT_0x" << GuestAddr << "_" << HostAddr << std::endl;
|
||||
fwrite(String.str().c_str(), 1, String.str().size(), fp);
|
||||
fmt::print(fp.get(), "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(void *HostAddr, uint32_t CodeSize, std::string const &Name) {
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
std::stringstream String;
|
||||
String << std::hex << HostAddr << " " << CodeSize << " " << Name << "_" << HostAddr << std::endl;
|
||||
fwrite(String.str().c_str(), 1, String.str().size(), fp);
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+11
-4
@@ -1,17 +1,24 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
#include <memory>
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore {
|
||||
class JITSymbols final {
|
||||
public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
void Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(void *HostAddr, uint32_t CodeSize, std::string const &Name);
|
||||
|
||||
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
|
||||
private:
|
||||
FILE* fp{};
|
||||
using FILEPtr = std::unique_ptr<FILE, decltype(&std::fclose)>;
|
||||
|
||||
FILEPtr fp;
|
||||
};
|
||||
}
|
||||
-13
@@ -1,13 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
static inline uint64_t AlignUp(uint64_t value, uint64_t size) {
|
||||
return value + (size - value % size) % size;
|
||||
};
|
||||
|
||||
static inline uint64_t AlignDown(uint64_t value, uint64_t size) {
|
||||
return value - value % size;
|
||||
};
|
||||
|
||||
|
||||
-41
@@ -1,41 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <iostream>
|
||||
#include <string.h>
|
||||
|
||||
class NetStream : public std::iostream {
|
||||
public:
|
||||
NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
|
||||
virtual ~NetStream();
|
||||
|
||||
private:
|
||||
class NetBuf : public std::streambuf {
|
||||
|
||||
public:
|
||||
NetBuf(int socketfd) {
|
||||
socket = socketfd;
|
||||
reset_output_buffer();
|
||||
}
|
||||
virtual ~NetBuf();
|
||||
|
||||
protected:
|
||||
virtual std::streamsize xsputn(const char* buffer, std::streamsize size);
|
||||
|
||||
virtual std::streambuf::int_type underflow();
|
||||
virtual std::streambuf::int_type overflow(std::streambuf::int_type ch);
|
||||
virtual int sync();
|
||||
|
||||
private:
|
||||
void reset_output_buffer() {
|
||||
// we always leave room for one extra char
|
||||
setp(std::begin(output_buffer), std::end(output_buffer) -1);
|
||||
}
|
||||
|
||||
int flushBuffer(const char *buffer, size_t size);
|
||||
|
||||
int socket;
|
||||
std::array<char, 1400> output_buffer;
|
||||
std::array<char, 1500> input_buffer; // enough for a typical packet
|
||||
};
|
||||
};
|
||||
+34
-2
@@ -3,12 +3,44 @@
|
||||
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <sys/stat.h>
|
||||
#include <memory>
|
||||
#include <pwd.h>
|
||||
#include <system_error>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
std::unique_ptr<std::string> CachePath;
|
||||
std::unique_ptr<std::string> EntryCache;
|
||||
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
return passwd->pw_dir;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const char *GetHomeDirectory() {
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
}
|
||||
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
return HomeDir;
|
||||
}
|
||||
|
||||
void InitializePaths() {
|
||||
CachePath = std::make_unique<std::string>();
|
||||
EntryCache = std::make_unique<std::string>();
|
||||
@@ -40,7 +72,7 @@ namespace FEXCore::Paths {
|
||||
// Ensure the folder structure is created for our Data
|
||||
if (!std::filesystem::exists(*EntryCache, ec) &&
|
||||
!std::filesystem::create_directories(*EntryCache, ec)) {
|
||||
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache->c_str());
|
||||
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+3
@@ -4,6 +4,9 @@
|
||||
namespace FEXCore::Paths {
|
||||
void InitializePaths();
|
||||
void ShutdownPaths();
|
||||
|
||||
const char *GetHomeDirectory();
|
||||
|
||||
std::string GetCachePath();
|
||||
std::string GetEntryCachePath();
|
||||
}
|
||||
+297
-15
@@ -16,9 +16,19 @@ extern "C" {
|
||||
|
||||
struct X80SoftFloat {
|
||||
#ifdef _M_X86_64
|
||||
// Define this to push some operations to x87
|
||||
// Only useful to see if precision loss is killing something
|
||||
// #define DEBUG_X86_FLOAT
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
#define BIGFLOAT long double
|
||||
#define BIGFLOATSIZE 10
|
||||
#else
|
||||
#define BIGFLOAT __float128
|
||||
#define BIGFLOATSIZE 16
|
||||
#endif
|
||||
#elif defined(_M_ARM_64)
|
||||
#define BIGFLOAT long double
|
||||
#define BIGFLOATSIZE 16
|
||||
#else
|
||||
#error No 128bit float for this target!
|
||||
#endif
|
||||
@@ -47,35 +57,131 @@ struct X80SoftFloat {
|
||||
|
||||
// Ops
|
||||
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
faddp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_add(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fsubp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sub(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fmulp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_mul(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fdivp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_div(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
X80SoftFloat Rem = extF80_rem(lhs, rhs);
|
||||
if (SignBit(Rem)) {
|
||||
Rem = extF80_add(Rem, rhs);
|
||||
}
|
||||
else {
|
||||
Rem.Sign = SignBit(lhs);
|
||||
}
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Rem;
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem1;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
|
||||
@@ -83,15 +189,47 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
|
||||
@@ -101,62 +239,190 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE("x87: Application used FSCALE which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fscale; # st0 = st0 * 2^(rdint(st1))
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs);
|
||||
BIGFLOAT Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE("x87: Application used F2XM1 which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
f2xm1; # st0 = 2^st(0) - 1
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE("x87: Application used FYL2X which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st(1)
|
||||
fldt %[lhs]; # st(0)
|
||||
fyl2x; # st(1) * log2l(st(0))
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE("x87: Application used FATAN which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs];
|
||||
fldt %[rhs];
|
||||
fpatan;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE("x87: Application used FTAN which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fptan;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE("x87: Application used FSIN which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsin;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE("x87: Application used FCOS which may have accuracy problems");
|
||||
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fcos;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsqrt;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sqrt(lhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
operator float() const {
|
||||
@@ -170,8 +436,14 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
operator BIGFLOAT() const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result{};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
return result;
|
||||
#endif
|
||||
}
|
||||
|
||||
operator int16_t() const {
|
||||
@@ -217,6 +489,12 @@ struct X80SoftFloat {
|
||||
*this = ui64_to_extF80(rhs);
|
||||
}
|
||||
|
||||
#if BIGFLOATSIZE == 10
|
||||
void operator=(const long double rhs) {
|
||||
memcpy(this, &rhs, sizeof(rhs));
|
||||
}
|
||||
#endif
|
||||
|
||||
operator void*() {
|
||||
return reinterpret_cast<void*>(this);
|
||||
}
|
||||
@@ -236,7 +514,11 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
X80SoftFloat(BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
|
||||
#else
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
X80SoftFloat(const int16_t rhs) {
|
||||
|
||||
+356
-51
@@ -1,43 +1,126 @@
|
||||
#include "Common/StringConv.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Common/Paths.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <pwd.h>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <list>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/sysinfo.h>
|
||||
#include <unistd.h>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Config {
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
return passwd->pw_dir;
|
||||
}
|
||||
return nullptr;
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
std::unique_ptr<std::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = std::make_unique<std::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
const char *GetHomeDirectory() {
|
||||
char const *HomeDir = getenv("HOME");
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
std::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
LogMan::Msg::EFmt("Couldn't get config list");
|
||||
return;
|
||||
}
|
||||
|
||||
return HomeDir;
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::string GetDataDirectory() {
|
||||
std::string DataDir{};
|
||||
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *DataXDG = getenv("XDG_DATA_HOME");
|
||||
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
if (DataOverride) {
|
||||
// Data override will override the complete directory
|
||||
DataDir = DataOverride;
|
||||
}
|
||||
else {
|
||||
DataDir = DataXDG ?: HomeDir;
|
||||
DataDir += "/.fex-emu/";
|
||||
}
|
||||
return DataDir;
|
||||
}
|
||||
|
||||
std::string GetConfigDirectory(bool Global) {
|
||||
@@ -46,7 +129,7 @@ namespace FEXCore::Config {
|
||||
ConfigDir = GLOBAL_DATA_DIRECTORY;
|
||||
}
|
||||
else {
|
||||
char const *HomeDir = GetHomeDirectory();
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
|
||||
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (ConfigOverride) {
|
||||
@@ -90,7 +173,7 @@ namespace FEXCore::Config {
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigFile.c_str());
|
||||
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
|
||||
// Let's go local in this case
|
||||
return "./" + Filename + ".json";
|
||||
}
|
||||
@@ -109,23 +192,6 @@ namespace FEXCore::Config {
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
std::string GetDataDirectory() {
|
||||
std::string DataDir{};
|
||||
|
||||
char const *HomeDir = GetHomeDirectory();
|
||||
char const *DataXDG = getenv("XDG_DATA_HOME");
|
||||
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
if (DataOverride) {
|
||||
// Data override will override the complete directory
|
||||
DataDir = DataOverride;
|
||||
}
|
||||
else {
|
||||
DataDir = DataXDG ?: HomeDir;
|
||||
DataDir += "/.fex-emu/";
|
||||
}
|
||||
return DataDir;
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
|
||||
}
|
||||
|
||||
@@ -225,7 +291,8 @@ namespace FEXCore::Config {
|
||||
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto &it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
|
||||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
}
|
||||
else {
|
||||
@@ -253,7 +320,7 @@ namespace FEXCore::Config {
|
||||
}
|
||||
}
|
||||
|
||||
std::string ExpandPath(std::string PathName) {
|
||||
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
@@ -279,6 +346,69 @@ namespace FEXCore::Config {
|
||||
return Path;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If the containerprefix and pathname isn't empty
|
||||
// Then we check if the pathname exists in our current namespace
|
||||
// If the path DOESN'T exist but DOES exist with the prefix applied
|
||||
// then redirect to the prefix
|
||||
//
|
||||
// This might not be expected behaviour for some edge cases but since
|
||||
// all paths aren't mounted inside the container, then it'll be fine
|
||||
//
|
||||
// Main catch case for this is the default thunk install folders
|
||||
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
|
||||
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
|
||||
if (!ContainerPrefix.empty() && !PathName.empty()) {
|
||||
if (!std::filesystem::exists(PathName)) {
|
||||
auto ContainerPath = ContainerPrefix + PathName;
|
||||
if (std::filesystem::exists(ContainerPath)) {
|
||||
return ContainerPath;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string ltrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string rtrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string trim(std::string String) {
|
||||
return rtrim(ltrim(String));
|
||||
}
|
||||
|
||||
|
||||
std::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
return "/run/host/";
|
||||
}
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -286,7 +416,9 @@ namespace FEXCore::Config {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THREADS)) {
|
||||
{
|
||||
// Always fix up the number of threads and create the configuration
|
||||
// Otherwise the application could receive zero as the number of threads
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
@@ -294,8 +426,28 @@ namespace FEXCore::Config {
|
||||
}
|
||||
}
|
||||
|
||||
auto ExpandPathIfExists = [](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(PathName);
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 2;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#endif
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
constexpr uint32_t MinCoreNumber = 0;
|
||||
#else
|
||||
constexpr uint32_t MinCoreNumber = 1;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
@@ -303,7 +455,7 @@ namespace FEXCore::Config {
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(PathName());
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
@@ -327,7 +479,7 @@ namespace FEXCore::Config {
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(PathName());
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
@@ -370,7 +522,7 @@ namespace FEXCore::Config {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string Data) {
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
|
||||
@@ -378,7 +530,7 @@ namespace FEXCore::Config {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string Data) {
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
@@ -417,6 +569,17 @@ namespace FEXCore::Config {
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
}
|
||||
else {
|
||||
return std::string(Default);
|
||||
}
|
||||
}
|
||||
|
||||
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
|
||||
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
|
||||
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
|
||||
@@ -442,5 +605,147 @@ namespace FEXCore::Config {
|
||||
}
|
||||
}
|
||||
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader();
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, bool Global);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader()
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation()} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, bool Global)
|
||||
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
std::unordered_map<std::string_view, std::string_view> EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (std::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key]=Value;
|
||||
}
|
||||
|
||||
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(it.first)).has_value()) {
|
||||
Set(it.second, std::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>();
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
+56
-4
@@ -32,7 +32,7 @@
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "1",
|
||||
"Default": "0",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
@@ -58,7 +58,7 @@
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
@@ -66,7 +66,7 @@
|
||||
},
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks/",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
@@ -94,6 +94,16 @@
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
},
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "H",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
@@ -137,6 +147,13 @@
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
},
|
||||
"SRA": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Set to false to disable Static Register Allocation"
|
||||
]
|
||||
},
|
||||
"Force32BitAllocator": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -146,8 +163,34 @@
|
||||
"Potentially useful for debugging memory problems",
|
||||
"32-bit allocator is always used if your host kernel is older than 4.17"
|
||||
]
|
||||
},
|
||||
"GlobalJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name all JIT state as one symbol",
|
||||
"Useful for querying how much time is spent inside of the JIT",
|
||||
"Profiling tools will show JIT time as FEXJIT"
|
||||
]
|
||||
},
|
||||
"LibraryJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols grouped by library",
|
||||
"Useful for querying how much time is spent in each guest library",
|
||||
"Can be used to help guide thunk generation"
|
||||
]
|
||||
},
|
||||
"BlockJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols",
|
||||
"Useful for determining hot blocks of code",
|
||||
"Has some file writing overhead per JIT block"
|
||||
]
|
||||
}
|
||||
|
||||
},
|
||||
"Logging": {
|
||||
"SilentLog": {
|
||||
@@ -158,6 +201,15 @@
|
||||
"Disables logging"
|
||||
]
|
||||
},
|
||||
"OutputSocket": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Socket to connect to",
|
||||
"eg: localhost:8087",
|
||||
"If set will override the OutputLog location"
|
||||
]
|
||||
},
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "stderr",
|
||||
+38
-17
@@ -4,9 +4,18 @@
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
class SyscallVisitor;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
@@ -34,7 +43,7 @@ namespace FEXCore::Context {
|
||||
delete CTX;
|
||||
}
|
||||
|
||||
bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
|
||||
return CTX->InitCore(Loader);
|
||||
}
|
||||
|
||||
@@ -42,7 +51,7 @@ namespace FEXCore::Context {
|
||||
CTX->CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
ExitHandler GetExitHandler(FEXCore::Context::Context *CTX) {
|
||||
ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->CustomExitHandler;
|
||||
}
|
||||
|
||||
@@ -62,23 +71,23 @@ namespace FEXCore::Context {
|
||||
return CTX->RunUntilExit();
|
||||
}
|
||||
|
||||
int GetProgramStatus(FEXCore::Context::Context *CTX) {
|
||||
int GetProgramStatus(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->GetProgramStatus();
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason GetExitReason(FEXCore::Context::Context *CTX) {
|
||||
FEXCore::Context::ExitReason GetExitReason(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
bool IsDone(FEXCore::Context::Context *CTX) {
|
||||
bool IsDone(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->IsPaused();
|
||||
}
|
||||
|
||||
void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
void GetCPUState(const FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
memcpy(State, CTX->ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, const FEXCore::Core::CPUState *State) {
|
||||
memcpy(CTX->ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
@@ -101,22 +110,26 @@ namespace FEXCore::Context {
|
||||
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
|
||||
}
|
||||
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
CTX->HandleCallback(RIP);
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
CTX->HandleCallback(Thread, RIP);
|
||||
}
|
||||
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterHostSignalHandler(Signal, Func, Required);
|
||||
CTX->RegisterHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
return CTX->CreateThread(NewThreadState, ParentTID);
|
||||
}
|
||||
|
||||
void ExecutionThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return CTX->ExecutionThread(Thread);
|
||||
}
|
||||
|
||||
void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return CTX->InitializeThread(Thread);
|
||||
}
|
||||
@@ -149,12 +162,20 @@ namespace FEXCore::Context {
|
||||
return CTX->CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
|
||||
CTX->AOTIRLoader = CacheReader;
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CTX->CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
CTX->AOTIRWriter = CacheWriter;
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
|
||||
CTX->SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
CTX->SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer) {
|
||||
CTX->SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
|
||||
|
||||
+113
-86
@@ -2,47 +2,49 @@
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <ostream>
|
||||
#include <set>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <string>
|
||||
#include <unordered_map>
|
||||
#include <queue>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class BlockSamplingData;
|
||||
class GdbServer;
|
||||
class SiganlDelegator;
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
}
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
@@ -56,38 +58,6 @@ namespace FEXCore::Context {
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
|
||||
/* RAData followed by IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData *GetRAData();
|
||||
IR::IRListView *GetIRData();
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndexEntry {
|
||||
uint64_t GuestStart;
|
||||
uint64_t DataOffset;
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndex {
|
||||
uint64_t Count;
|
||||
uint64_t DataBase;
|
||||
AOTIRInlineIndexEntry Entries[0];
|
||||
|
||||
AOTIRInlineEntry *Find(uint64_t GuestStart);
|
||||
AOTIRInlineEntry *GetInlineEntry(uint64_t DataOffset);
|
||||
};
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
std::unique_ptr<std::ostream> Stream;
|
||||
std::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData);
|
||||
};
|
||||
|
||||
struct Context {
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
@@ -121,7 +91,13 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(DumpIR, DUMPIR);
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
@@ -149,30 +125,6 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex *Array;
|
||||
void *mapping;
|
||||
size_t size;
|
||||
};
|
||||
|
||||
std::unordered_map<std::string, AOTIRCacheEntry> AOTIRCache;
|
||||
std::function<int(const std::string&)> AOTIRLoader;
|
||||
std::function<std::unique_ptr<std::ostream>(const std::string&)> AOTIRWriter;
|
||||
std::unordered_map<std::string, AOTIRCaptureCacheEntry> AOTIRCaptureCache;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
uint64_t Len;
|
||||
uint64_t Offset;
|
||||
std::string fileid;
|
||||
std::string filename;
|
||||
void *CachedFileEntry;
|
||||
bool ContainsCode;
|
||||
};
|
||||
|
||||
std::map<uint64_t, AddrToFileEntry> AddrToFile;
|
||||
std::map<std::string, std::string> FilesWithCode;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
@@ -183,7 +135,7 @@ namespace FEXCore::Context {
|
||||
Context();
|
||||
~Context();
|
||||
|
||||
bool InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus() const;
|
||||
bool IsPaused() const { return !Running; }
|
||||
@@ -199,7 +151,7 @@ namespace FEXCore::Context {
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
void HandleCallback(uint64_t RIP);
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
@@ -240,23 +192,77 @@ namespace FEXCore::Context {
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
bool LoadAOTIRCache(int streamfd);
|
||||
void FinalizeAOTIRCache();
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
* @param CompileThread Is this for the compile service or not?
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
* This is exposed because the CompileService needs to initialize compilers while copying data from
|
||||
* the paired InternalThreadState that it is compiling code for
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* The OS thread will wait until RunThread is executed
|
||||
*/
|
||||
void InitializeThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
/**
|
||||
* @brief Starts the OS thread object to start executing guest code
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void RunThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
|
||||
|
||||
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
@@ -266,17 +272,44 @@ namespace FEXCore::Context {
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
|
||||
#if ENABLE_JITSYMBOLS
|
||||
FEXCore::JITSymbols Symbols;
|
||||
#endif
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void FinalizeAOTIRCache() {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* InitCore and CreateThread both call this to finish up thread object initialization
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
@@ -289,13 +322,7 @@ namespace FEXCore::Context {
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
std::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
|
||||
bool StartPaused = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
+594
-216
@@ -2,13 +2,22 @@
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <signal.h>
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock16B, TYPE_16BYTE_SPLIT);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas16Tear, TYPE_CAS_16BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas32Tear, TYPE_CAS_32BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas64Tear, TYPE_CAS_64BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas128Tear, TYPE_CAS_128BIT_TEAR);
|
||||
|
||||
static __uint128_t LoadAcquire128(uint64_t Addr) {
|
||||
__uint128_t Result{};
|
||||
uint64_t Lower;
|
||||
@@ -60,194 +69,6 @@ static bool StoreCAS8(uint8_t &Expected, uint8_t Val, uint64_t Addr) {
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
|
||||
uint32_t DesiredReg1 = Instr & 0b11111;
|
||||
uint32_t DesiredReg2 = DesiredReg1 + 1;
|
||||
uint32_t ExpectedReg1 = (Instr >> 16) & 0b11111;
|
||||
uint32_t ExpectedReg2 = ExpectedReg1 + 1;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
if (Size == 0) {
|
||||
// 32bit
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
uint32_t DesiredLower = mcontext->regs[DesiredReg1];
|
||||
uint32_t DesiredUpper = mcontext->regs[DesiredReg2];
|
||||
|
||||
uint32_t ExpectedLower = mcontext->regs[ExpectedReg1];
|
||||
uint32_t ExpectedUpper = mcontext->regs[ExpectedReg2];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
// Intel will do a "split lock" which locks the full bus
|
||||
// AMD will tear instead
|
||||
// Both cross-cacheline and cross 16byte both need dual CAS loops that can tear
|
||||
// ARMv8.4 LSE2 solves all atomic issues except cross-cacheline
|
||||
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
uint64_t AddrUpper = Addr + 8;
|
||||
|
||||
// Crosses a 16byte boundary
|
||||
// Need to do 256bit atomic, but since that doesn't exist we need to do a dual CAS loop
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = DesiredUpper;
|
||||
Desired <<= 32;
|
||||
Desired |= DesiredLower;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = ExpectedUpper;
|
||||
Expected <<= 32;
|
||||
Expected |= ExpectedLower;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
__uint128_t LoadOrderUpper = LoadAcquire64(AddrUpper);
|
||||
LoadOrderUpper <<= 64;
|
||||
__uint128_t TmpActual = LoadOrderUpper | LoadAcquire64(Addr);
|
||||
|
||||
// Set up expected
|
||||
TmpExpected = TmpActual;
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
uint64_t TmpExpectedLower = TmpExpected;
|
||||
uint64_t TmpExpectedUpper = TmpExpected >> 64;
|
||||
|
||||
uint64_t TmpDesiredLower = TmpDesired;
|
||||
uint64_t TmpDesiredUpper = TmpDesired >> 64;
|
||||
|
||||
if (TmpExpected == TmpActual) {
|
||||
if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
}
|
||||
}
|
||||
|
||||
TmpExpected = TmpExpectedUpper;
|
||||
TmpExpected <<= 64;
|
||||
TmpExpected |= TmpExpectedLower;
|
||||
}
|
||||
else {
|
||||
// Mismatch up front
|
||||
TmpExpected = TmpActual;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t> *Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = (uint64_t)DesiredUpper << 32 | DesiredLower;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = (uint64_t)ExpectedUpper << 32 | ExpectedLower;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t DoLoad16(uint64_t Addr) {
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
@@ -417,6 +238,346 @@ std::pair<uint64_t, uint64_t> DoLoad128(uint64_t Addr) {
|
||||
return {ResultLower, ResultUpper};
|
||||
}
|
||||
|
||||
static bool RunCASPAL(void *_ucontext, void *_info, uint32_t Size, uint32_t DesiredReg1, uint32_t DesiredReg2, uint32_t ExpectedReg1, uint32_t ExpectedReg2, uint32_t AddressReg) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
|
||||
//Bus_ADRALN check happens in HandleCASPAL and HandleCASPAL_ARMv8
|
||||
|
||||
if (Size == 0) {
|
||||
// 32bit
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
uint32_t DesiredLower = mcontext->regs[DesiredReg1];
|
||||
uint32_t DesiredUpper = mcontext->regs[DesiredReg2];
|
||||
|
||||
uint32_t ExpectedLower = mcontext->regs[ExpectedReg1];
|
||||
uint32_t ExpectedUpper = mcontext->regs[ExpectedReg2];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
// Intel will do a "split lock" which locks the full bus
|
||||
// AMD will tear instead
|
||||
// Both cross-cacheline and cross 16byte both need dual CAS loops that can tear
|
||||
// ARMv8.4 LSE2 solves all atomic issues except cross-cacheline
|
||||
|
||||
// Check for Split lock across a cacheline
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
}
|
||||
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
uint64_t AddrUpper = Addr + 8;
|
||||
|
||||
// Crosses a 16byte boundary
|
||||
// Need to do 256bit atomic, but since that doesn't exist we need to do a dual CAS loop
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = DesiredUpper;
|
||||
Desired <<= 32;
|
||||
Desired |= DesiredLower;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = ExpectedUpper;
|
||||
Expected <<= 32;
|
||||
Expected |= ExpectedLower;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
__uint128_t LoadOrderUpper = LoadAcquire64(AddrUpper);
|
||||
LoadOrderUpper <<= 64;
|
||||
__uint128_t TmpActual = LoadOrderUpper | LoadAcquire64(Addr);
|
||||
|
||||
// Set up expected
|
||||
TmpExpected = TmpActual;
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
uint64_t TmpExpectedLower = TmpExpected;
|
||||
uint64_t TmpExpectedUpper = TmpExpected >> 64;
|
||||
|
||||
uint64_t TmpDesiredLower = TmpDesired;
|
||||
uint64_t TmpDesiredUpper = TmpDesired >> 64;
|
||||
|
||||
if (TmpExpected == TmpActual) {
|
||||
if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
FEXCORE_TELEMETRY_SET(Cas128Tear, 1);
|
||||
}
|
||||
}
|
||||
|
||||
TmpExpected = TmpExpectedUpper;
|
||||
TmpExpected <<= 64;
|
||||
TmpExpected |= TmpExpectedLower;
|
||||
}
|
||||
else {
|
||||
// Mismatch up front
|
||||
TmpExpected = TmpActual;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t> *Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = (uint64_t)DesiredUpper << 32 | DesiredLower;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = (uint64_t)ExpectedUpper << 32 | ExpectedLower;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
|
||||
uint32_t DesiredReg1 = Instr & 0b11111;
|
||||
uint32_t DesiredReg2 = DesiredReg1 + 1;
|
||||
uint32_t ExpectedReg1 = (Instr >> 16) & 0b11111;
|
||||
uint32_t ExpectedReg2 = ExpectedReg1 + 1;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
return RunCASPAL(_ucontext, _info, Size, DesiredReg1, DesiredReg2, ExpectedReg1, ExpectedReg2, AddressReg);
|
||||
}
|
||||
|
||||
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return 0;
|
||||
}
|
||||
|
||||
// caspair
|
||||
// [1] ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc)); <-- DataReg & AddrReg
|
||||
// [2] cmp(TMP2.W(), Expected.first.W()); <-- ExpectedReg1
|
||||
// [3] ccmp(TMP3.W(), Expected.second.W(), NoFlag, Condition::eq); <-- ExpectedREg2
|
||||
// [4] b(&LoopNotExpected, Condition::ne);
|
||||
// [5] stlxp(TMP2.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc)); <-- DesiredReg
|
||||
// [6] cbnz(TMP2.W(), &LoopTop);
|
||||
// [7] mov(Dst.first.W(), Expected.first.W());
|
||||
// [8] mov(Dst.second.W(), Expected.second.W());
|
||||
// [9] b(&LoopExpected);
|
||||
// [10] mov(Dst.first.W(), TMP2.W());
|
||||
// [11] mov(Dst.second.W(), TMP3.W());
|
||||
// [12] clrex();
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t DataReg2 = (Instr >> 10) & 0x1F;
|
||||
|
||||
uint32_t ExpectedReg1{};
|
||||
uint32_t ExpectedReg2{};
|
||||
|
||||
uint32_t DesiredReg1{};
|
||||
uint32_t DesiredReg2{};
|
||||
|
||||
if(Size == 1) {
|
||||
// 64-bit pair happens on paranoid vector loads
|
||||
// [1] ldaxp(TMP1, TMP2, MemSrc);
|
||||
// [2] clrex();
|
||||
//
|
||||
// 64-bit pair happens on paranoid vector stores
|
||||
// [1] ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS
|
||||
// [2] stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
|
||||
// [3] cbnz(TMP3, &B); // < Overwritten with DMB
|
||||
|
||||
if (DataReg == 31) {
|
||||
}
|
||||
else {
|
||||
uint32_t NextInstr = PC[1];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::CLREX_MASK) == FEXCore::ArchHelpers::Arm64::CLREX_INST) {
|
||||
uint64_t Addr = mcontext->regs[AddrReg];
|
||||
|
||||
auto Res = DoLoad128(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (DataReg != 31) {
|
||||
mcontext->regs[DataReg] = std::get<0>(Res);
|
||||
}
|
||||
if (DataReg2 != 31) {
|
||||
mcontext->regs[DataReg2] = std::get<1>(Res);
|
||||
}
|
||||
|
||||
// Skip ldaxp and clrex
|
||||
return 2 * sizeof(uint32_t);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
//Only 32-bit pairs
|
||||
for(int i = 1; i < 10; i++) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
ExpectedReg1 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::CCMP_MASK) == FEXCore::ArchHelpers::Arm64::CCMP_INST) {
|
||||
ExpectedReg2 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXP_MASK) == FEXCore::ArchHelpers::Arm64::STLXP_INST) {
|
||||
DesiredReg1 = (NextInstr & 0x1F);
|
||||
DesiredReg2 = (NextInstr >> 10) & 0x1F;
|
||||
}
|
||||
}
|
||||
|
||||
//mov expected into the temp registers used by JIT
|
||||
mcontext->regs[DataReg] = mcontext->regs[ExpectedReg1];
|
||||
mcontext->regs[DataReg2] = mcontext->regs[ExpectedReg2];
|
||||
|
||||
if(RunCASPAL(_ucontext, _info, Size, DesiredReg1, DesiredReg2, DataReg, DataReg2, AddrReg)) {
|
||||
return 9 * sizeof(uint32_t); // skip to mov + clrex
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t DataReg2 = (Instr >> 10) & 0x1F;
|
||||
|
||||
if(Size == 1) {
|
||||
// 64-bit pair happens on paranoid vector stores
|
||||
// [0] ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
// [1] stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
|
||||
// [2] cbnz(TMP3, &B); // < Overwritten with DMB
|
||||
if (DataReg == 31) {
|
||||
uint32_t NextInstr = PC[1];
|
||||
AddrReg = (NextInstr >> 5) & 0x1F;
|
||||
DataReg = NextInstr & 0x1F;
|
||||
DataReg2 = (NextInstr >> 10) & 0x1F;
|
||||
uint32_t STP =
|
||||
(0b10 << 30) |
|
||||
(0b101001000000000 << 15) |
|
||||
(DataReg2 << 10) |
|
||||
(AddrReg << 5) |
|
||||
DataReg;
|
||||
|
||||
PC[0] = DMB;
|
||||
PC[1] = STP;
|
||||
PC[2] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[0], 16);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
using CASExpectedFn = T (*)(T Src, T Expected);
|
||||
template <typename T>
|
||||
@@ -430,9 +591,16 @@ uint16_t DoCAS16(
|
||||
uint64_t Addr,
|
||||
CASExpectedFn<uint16_t> ExpectedFunction,
|
||||
CASDesiredFn<uint16_t> DesiredFunction) {
|
||||
|
||||
if ((Addr & 63) == 63) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
}
|
||||
|
||||
// 16 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
|
||||
// Address crosses over 16byte or 64byte threshold
|
||||
// Need a dual 8bit CAS loop
|
||||
uint64_t AddrUpper = Addr + 1;
|
||||
@@ -468,6 +636,7 @@ uint16_t DoCAS16(
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas16Tear, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -706,9 +875,16 @@ uint32_t DoCAS32(
|
||||
uint64_t Addr,
|
||||
CASExpectedFn<uint32_t> ExpectedFunction,
|
||||
CASDesiredFn<uint32_t> DesiredFunction) {
|
||||
|
||||
if ((Addr & 63) > 60) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
}
|
||||
|
||||
// 32 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 12) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
|
||||
// Address crosses over 16byte threshold
|
||||
// Needs dual 4 byte CAS loop
|
||||
uint64_t Alignment = Addr & 0b11;
|
||||
@@ -754,6 +930,7 @@ uint32_t DoCAS32(
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas32Tear, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -936,9 +1113,16 @@ uint64_t DoCAS64(
|
||||
uint64_t Addr,
|
||||
CASExpectedFn<uint64_t> ExpectedFunction,
|
||||
CASDesiredFn<uint64_t> DesiredFunction) {
|
||||
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
}
|
||||
|
||||
// 64bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
uint64_t AddrUpper = Addr + 8;
|
||||
@@ -986,6 +1170,7 @@ uint64_t DoCAS64(
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas64Tear, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1091,20 +1276,8 @@ uint64_t DoCAS64(
|
||||
}
|
||||
}
|
||||
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
static bool RunCASAL(void *_ucontext, void *_info, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DesiredReg = Instr & 0b11111;
|
||||
uint32_t ExpectedReg = (Instr >> 16) & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
@@ -1185,6 +1358,22 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DesiredReg = Instr & 0b11111;
|
||||
uint32_t ExpectedReg = (Instr >> 16) & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
return RunCASAL(_ucontext, _info, Size, DesiredReg, ExpectedReg, AddressReg);
|
||||
}
|
||||
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
@@ -1247,9 +1436,8 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
DesiredFunction = SWAPDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x", Op);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
|
||||
auto Res = DoCAS16<true>(
|
||||
@@ -1309,9 +1497,8 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
DesiredFunction = SWAPDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x", Op);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
|
||||
auto Res = DoCAS32<true>(
|
||||
@@ -1371,9 +1558,8 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
DesiredFunction = SWAPDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x", Op);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
|
||||
auto Res = DoCAS64<true>(
|
||||
@@ -1527,6 +1713,52 @@ bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static uint64_t HandleCAS_NoAtomics(void *_ucontext, void *_info)
|
||||
{
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
|
||||
// ARMv8.0 CAS
|
||||
// [1] ldaxrb(TMP2.W(), MemOperand(MemSrc))
|
||||
// [2] cmp (TMP2.W(), Expected.W())
|
||||
// [3] b
|
||||
// [4] stlxrb(TMP3.W(), Desired.W(), MemOperand(MemSrc)
|
||||
// [5] cbnz
|
||||
// [6] mov
|
||||
// [7] b
|
||||
// [8] mov (.., TMP2.W());
|
||||
// [9] clrex
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
uint32_t AddressReg = GetRnReg(Instr);
|
||||
uint32_t ResultReg = GetRdReg(Instr); //TMP2
|
||||
uint32_t DesiredReg = 0;
|
||||
uint32_t ExpectedReg = 0;
|
||||
for (size_t i = 1; i < 6; ++i) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
const uint32_t StoreAddressReg = GetRnReg(NextInstr);
|
||||
LOGMAN_THROW_A_FMT(StoreAddressReg == AddressReg, "StoreExclusive memory register didn't match the store exclusive register");
|
||||
#endif
|
||||
DesiredReg = GetRdReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
ExpectedReg = GetRmReg(NextInstr);
|
||||
}
|
||||
}
|
||||
//set up CASAL by doing mov(TMP2, Expected)
|
||||
mcontext->regs[ResultReg] = mcontext->regs[ExpectedReg];
|
||||
|
||||
if(RunCASAL(_ucontext, _info, Size, DesiredReg, ResultReg, AddressReg)) {
|
||||
return 7 * sizeof(uint32_t); //jump to mov to allocated register
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
@@ -1611,6 +1843,9 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::AND_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_AND;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
@@ -1626,8 +1861,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
uint32_t StoreAddressReg = GetRnReg(NextInstr);
|
||||
LOGMAN_THROW_A(StoreAddressReg == AddressReg, "StoreExclusive memory register didn't match the store exclusive register");
|
||||
const uint32_t StoreAddressReg = GetRnReg(NextInstr);
|
||||
LOGMAN_THROW_A_FMT(StoreAddressReg == AddressReg, "StoreExclusive memory register didn't match the store exclusive register");
|
||||
#endif
|
||||
uint32_t StatusReg = GetRmReg(NextInstr);
|
||||
uint32_t StoreResultReg = GetRdReg(NextInstr);
|
||||
@@ -1646,7 +1881,7 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
break;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::A("Unknown instruction 0x%08x", NextInstr);
|
||||
LogMan::Msg::AFmt("Unknown instruction 0x{:08x}", NextInstr);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1712,9 +1947,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
|
||||
auto Res = DoCAS16<DoRetry>(
|
||||
@@ -1789,9 +2023,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
|
||||
auto Res = DoCAS32<DoRetry>(
|
||||
@@ -1866,9 +2099,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
|
||||
auto Res = DoCAS64<DoRetry>(
|
||||
@@ -1888,4 +2120,150 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
return NumInstructionsToSkip * 4;
|
||||
}
|
||||
|
||||
bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
#ifdef _M_ARM_64
|
||||
constexpr bool is_arm64 = true;
|
||||
#else
|
||||
constexpr bool is_arm64 = false;
|
||||
#endif
|
||||
|
||||
if constexpr (is_arm64) {
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
// 1 = 16bit
|
||||
// 2 = 32bit
|
||||
// 3 = 64bit
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t LDR = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXP_MASK) == FEXCore::ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
//Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleCASPAL_ARMv8(ucontext, info, Instr);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + BytesToSkip);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicVectorStore(ucontext, info, Instr)) {
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::STLXP_MASK) == FEXCore::ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
//Should not trigger - middle of an LDAXP/STAXP pair.
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: {} Instruction: 0x{:08x}\n", Op, fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXR_MASK) == FEXCore::ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ucontext, info);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + BytesToSkip);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXR: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[-1], 16);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -30,9 +30,17 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
|
||||
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
|
||||
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
|
||||
|
||||
enum ExclusiveAtomicPairType {
|
||||
TYPE_SWAP,
|
||||
TYPE_ADD,
|
||||
@@ -60,6 +68,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t RN_OFFSET = 5;
|
||||
constexpr uint32_t RM_OFFSET = 16;
|
||||
|
||||
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1011'0000'0000; // Inner shareable all
|
||||
|
||||
inline uint32_t GetRdReg(uint32_t Instr) {
|
||||
return (Instr >> RD_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
@@ -77,6 +88,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr);
|
||||
[[nodiscard]] bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext);
|
||||
}
|
||||
@@ -1,45 +1,31 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "cpu-features.h"
|
||||
#include "aarch64/instructions-aarch64.h"
|
||||
#include "utils-vixl.h"
|
||||
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (ctx->HostFeatures.SupportsAtomics) {
|
||||
// Hypervisor can hide this on the c630?
|
||||
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
|
||||
}
|
||||
|
||||
SetCPUFeatures(Features);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile ("mrs %[ctr], ctr_el0"
|
||||
: [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
@@ -64,7 +50,7 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
MemOperand PairOffset(sp, -16, PreIndex);
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x19, x20},
|
||||
{x21, x22},
|
||||
{x23, x24},
|
||||
@@ -127,7 +113,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
MemOperand PairOffset(sp, 16, PostIndex);
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x29, x30},
|
||||
{x27, x28},
|
||||
{x25, x26},
|
||||
@@ -142,23 +128,53 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs() {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
stp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t SpillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & SpillMask) &&
|
||||
((1U << Reg2.GetCode()) & SpillMask)) {
|
||||
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & SpillMask)) {
|
||||
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & SpillMask)) {
|
||||
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs() {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
ldp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t FillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & FillMask) &&
|
||||
((1U << Reg2.GetCode()) & FillMask)) {
|
||||
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FillMask)) {
|
||||
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FillMask)) {
|
||||
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -222,7 +238,7 @@ void Arm64Emitter::ResetStack() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetBuffer()->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
|
||||
nop();
|
||||
}
|
||||
|
||||
@@ -1,7 +1,16 @@
|
||||
#pragma once
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "platform-vixl.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
@@ -49,15 +58,12 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(size_t size);
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
|
||||
vixl::aarch64::CPU CPU;
|
||||
bool SupportsAtomics{};
|
||||
bool SupportsRCPC{};
|
||||
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
void SpillStaticRegs();
|
||||
void FillStaticRegs();
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t SpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t FillMask = ~0U);
|
||||
|
||||
void PushDynamicRegsAndLR();
|
||||
void PopDynamicRegsAndLR();
|
||||
@@ -70,8 +76,7 @@ protected:
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
|
||||
uint32_t DCacheLineSize{};
|
||||
uint32_t ICacheLineSize{};
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
|
||||
@@ -11,16 +12,16 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
// Obvously such a configuration can't do the actual arm64-specific stuff
|
||||
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE("HandleCASPAL Not Implemented");
|
||||
ERROR_AND_DIE_FMT("HandleCASPAL Not Implemented");
|
||||
}
|
||||
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE("HandleCASAL Not Implemented");
|
||||
ERROR_AND_DIE_FMT("HandleCASAL Not Implemented");
|
||||
}
|
||||
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE("HandleAtomicMemOp Not Implemented");
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
#endif
|
||||
|
||||
}
|
||||
}
|
||||
+77
-22
@@ -13,16 +13,27 @@
|
||||
|
||||
namespace FEXCore::ArchHelpers::Context {
|
||||
|
||||
enum ContextFlags : uint32_t {
|
||||
CONTEXT_FLAG_INJIT = (1U << 0),
|
||||
CONTEXT_FLAG_32BIT = (1U << 1),
|
||||
};
|
||||
|
||||
struct X86ContextBackup {
|
||||
// Host State
|
||||
// RIP and RSP is stored in GPRs here
|
||||
uint64_t GPRs[23];
|
||||
FEXCore::x86_64::_libc_fpstate FPRState;
|
||||
uint64_t sa_mask;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
int Signal;
|
||||
uint32_t Flags;
|
||||
uint64_t OriginalRIP;
|
||||
uint64_t FPStateLocation;
|
||||
uint64_t UContextLocation;
|
||||
uint64_t SigInfoLocation;
|
||||
FEXCore::Core::CPUState GuestState;
|
||||
|
||||
static constexpr int RedZoneSize = 128;
|
||||
};
|
||||
|
||||
@@ -35,15 +46,27 @@ struct ArmContextBackup {
|
||||
uint32_t FPSR;
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
uint64_t sa_mask;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
int Signal;
|
||||
uint32_t Flags;
|
||||
uint64_t OriginalRIP;
|
||||
uint64_t FPStateLocation;
|
||||
uint64_t UContextLocation;
|
||||
uint64_t SigInfoLocation;
|
||||
FEXCore::Core::CPUState GuestState;
|
||||
|
||||
// Arm64 doesn't have a red zone
|
||||
static constexpr int RedZoneSize = 0;
|
||||
};
|
||||
|
||||
static inline ucontext_t* GetUContext(void* ucontext) {
|
||||
ucontext_t* _context = (ucontext_t*)ucontext;
|
||||
return _context;
|
||||
}
|
||||
|
||||
static inline mcontext_t* GetMContext(void* ucontext) {
|
||||
ucontext_t* _context = (ucontext_t*)ucontext;
|
||||
return &_context->uc_mcontext;
|
||||
@@ -52,6 +75,20 @@ static inline mcontext_t* GetMContext(void* ucontext) {
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
|
||||
constexpr uint32_t FPR_MAGIC = 0x46508001U;
|
||||
|
||||
struct HostCTXHeader {
|
||||
uint32_t Magic;
|
||||
uint32_t Size;
|
||||
};
|
||||
|
||||
struct HostFPRState {
|
||||
HostCTXHeader Head;
|
||||
uint32_t FPSR;
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
};
|
||||
|
||||
static inline uint64_t GetSp(void* ucontext) {
|
||||
return GetMContext(ucontext)->sp;
|
||||
}
|
||||
@@ -84,24 +121,19 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
GetMContext(ucontext)->regs[id] = val;
|
||||
}
|
||||
|
||||
constexpr uint32_t FPR_MAGIC = 0x46508001U;
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
|
||||
struct HostCTXHeader {
|
||||
uint32_t Magic;
|
||||
uint32_t Size;
|
||||
};
|
||||
|
||||
struct HostFPRState {
|
||||
HostCTXHeader Head;
|
||||
uint32_t FPSR;
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
};
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
|
||||
using ContextBackup = ArmContextBackup;
|
||||
template <typename T>
|
||||
static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, ArmContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
memcpy(&Backup->GPRs[0], &_mcontext->regs[0], 31 * sizeof(uint64_t));
|
||||
@@ -111,22 +143,27 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Host FPR state starts at _mcontext->reserved[0];
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
Backup->FPSR = HostState->FPSR;
|
||||
Backup->FPCR = HostState->FPCR;
|
||||
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
|
||||
// Save the signal mask so we can restore it
|
||||
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
|
||||
} else {
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, ArmContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
HostState->FPCR = Backup->FPCR;
|
||||
HostState->FPSR = Backup->FPSR;
|
||||
@@ -136,8 +173,12 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
ArchHelpers::Context::SetPc(ucontext, Backup->PrevPC);
|
||||
ArchHelpers::Context::SetSp(ucontext, Backup->PrevSP);
|
||||
memcpy(&_mcontext->regs[0], &Backup->GPRs[0], 31 * sizeof(uint64_t));
|
||||
|
||||
// Restore the signal mask now
|
||||
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
|
||||
} else {
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -170,17 +211,22 @@ static inline void SetState(void* ucontext, uint64_t val) {
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE("Not impelented for x86 host");
|
||||
ERROR_AND_DIE_FMT("Not impelented for x86 host");
|
||||
}
|
||||
|
||||
static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
ERROR_AND_DIE("Not impelented for x86 host");
|
||||
ERROR_AND_DIE_FMT("Not impelented for x86 host");
|
||||
}
|
||||
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
using ContextBackup = X86ContextBackup;
|
||||
template <typename T>
|
||||
static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, X86ContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
// Copy the GPRs
|
||||
@@ -188,25 +234,34 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
// Copy the FPRState
|
||||
memcpy(&Backup->FPRState, _mcontext->fpregs, sizeof(X86ContextBackup::FPRState));
|
||||
// XXX: Save 256bit and 512bit AVX register state
|
||||
|
||||
// Save the signal mask so we can restore it
|
||||
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
|
||||
} else {
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, X86ContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
// Copy the GPRs
|
||||
memcpy(&_mcontext->gregs[0], &Backup->GPRs[0], sizeof(X86ContextBackup::GPRs));
|
||||
// Copy the FPRState
|
||||
memcpy(_mcontext->fpregs, &Backup->FPRState, sizeof(X86ContextBackup::FPRState));
|
||||
|
||||
// Restore the signal mask now
|
||||
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
|
||||
} else {
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
} // namespace FEXCore::ArchHelpers::Context
|
||||
} // namespace FEXCore::ArchHelpers::Context
|
||||
@@ -2,6 +2,7 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore {
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
@@ -26,7 +27,7 @@ namespace FEXCore {
|
||||
<< std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::D("Dumped %d blocks of sampling data", SamplingMap.size());
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockSamplingData {
|
||||
|
||||
+412
-49
@@ -5,8 +5,16 @@ desc: Handles presented capability bits for guest cpu
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
|
||||
#include <cstring>
|
||||
@@ -15,6 +23,66 @@ $end_info$
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
namespace ProductNames {
|
||||
#ifdef _M_ARM_64
|
||||
static const char ARM_UNKNOWN[] = "Unknown ARM CPU";
|
||||
static const char ARM_A57[] = "Cortex-A57";
|
||||
static const char ARM_A72[] = "Cortex-A72";
|
||||
static const char ARM_A73[] = "Cortex-A73";
|
||||
static const char ARM_A75[] = "Cortex-A75";
|
||||
static const char ARM_A76[] = "Cortex-A76";
|
||||
static const char ARM_A76AE[] = "Cortex-A76AE";
|
||||
static const char ARM_V1[] = "Neoverse V1";
|
||||
static const char ARM_A77[] = "Cortex-A77";
|
||||
static const char ARM_A78[] = "Cortex-A78";
|
||||
static const char ARM_A78AE[] = "Cortex-A78AE";
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_E1[] = "Neoverse E1";
|
||||
static const char ARM_A35[] = "Cortex-A35";
|
||||
static const char ARM_A53[] = "Cortex-A53";
|
||||
static const char ARM_A55[] = "Cortex-A55";
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
static const char ARM_Kryo400[] = "Kryo 4xx/5xx";
|
||||
|
||||
static const char ARM_Kryo200S[] = "Kryo 2xx S";
|
||||
static const char ARM_Kryo300S[] = "Kryo 3xx S";
|
||||
static const char ARM_Kryo400S[] = "Kryo 4xx/5xx S";
|
||||
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
|
||||
static const char ARM_Firestorm[] = "Apple Firestorm";
|
||||
static const char ARM_Icestorm[] = "Apple Icestorm";
|
||||
#else
|
||||
static const char UNKNOWN[] = "Unknown CPU";
|
||||
#endif
|
||||
}
|
||||
|
||||
static uint32_t GetCPUID() {
|
||||
uint32_t CPU{};
|
||||
FHU::Syscalls::getcpu(&CPU, nullptr);
|
||||
return CPU;
|
||||
}
|
||||
|
||||
static uint32_t CalculateNumberOfCPUs() {
|
||||
size_t CPUs = 1;
|
||||
|
||||
while(std::filesystem::exists("/sys/devices/system/cpu/cpu" + std::to_string(CPUs))) {
|
||||
CPUs++;
|
||||
}
|
||||
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
@@ -42,6 +110,245 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
|
||||
uint64_t MIDR{};
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
std::error_code ec{};
|
||||
std::string MIDRPath = "/sys/devices/system/cpu/cpu" + std::to_string(i) + "/regs/identification/midr_el1";
|
||||
if (std::filesystem::exists(MIDRPath, ec)) {
|
||||
std::vector<char> Data{};
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFile(Data, MIDRPath, 18)) {
|
||||
uint64_t NewMIDR{};
|
||||
std::string_view MIDRView(&Data.at(0), 18);
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct CPUMIDR {
|
||||
uint8_t Implementer;
|
||||
uint16_t Part;
|
||||
bool DefaultBig; // Defaults to a big core
|
||||
const char *ProductName{};
|
||||
};
|
||||
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 35> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
|
||||
// Denver rated above A57 to match TX2 weirdness
|
||||
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
|
||||
|
||||
{0x41, 0xd07, 1, ProductNames::ARM_A57}, // A57
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
{0x41, 0xd05, 0, ProductNames::ARM_A55}, // A55
|
||||
{0x51, 0x805, 0, ProductNames::ARM_Kryo400S}, // Kryo 4xx/5xx Silver (A55 based)
|
||||
{0x51, 0x803, 0, ProductNames::ARM_Kryo300S}, // Kryo 3xx Silver (A55 based)
|
||||
{0x41, 0xd03, 0, ProductNames::ARM_A53}, // A53
|
||||
{0x51, 0x801, 0, ProductNames::ARM_Kryo200S}, // Kryo 2xx Silver (A53 based)
|
||||
{0x41, 0xd04, 0, ProductNames::ARM_A35}, // A35
|
||||
|
||||
{0x41, 0, 0, ProductNames::ARM_UNKNOWN}, // Invalid CPU or Apple CPU inside Parallels VM
|
||||
{0x0, 0, 0, ProductNames::ARM_UNKNOWN}, // Invalid starting point is lowest ranked
|
||||
}};
|
||||
|
||||
auto FindDefinedMIDR = [](uint32_t MIDR) -> const CPUMIDR* {
|
||||
uint8_t Implementer = MIDR >> 24;
|
||||
uint16_t Part = (MIDR >> 4) & 0xFFF;
|
||||
|
||||
for (auto &MIDROption : CPUMIDRs) {
|
||||
if (MIDROption.Implementer == Implementer &&
|
||||
MIDROption.Part == Part) {
|
||||
return &MIDROption;
|
||||
}
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
};
|
||||
|
||||
if (Hybrid) {
|
||||
// Walk the MIDRs and calculate big little designs
|
||||
std::vector<const CPUMIDR*> BigCores;
|
||||
std::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
if (MIDROption) {
|
||||
// Found one
|
||||
if (MIDROption->DefaultBig) {
|
||||
BigCores.emplace_back(MIDROption);
|
||||
}
|
||||
else {
|
||||
LittleCores.emplace_back(MIDROption);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If we didn't insert this MIDR then claim it is a little core.
|
||||
LittleCores.emplace_back(&CPUMIDRs.back());
|
||||
}
|
||||
}
|
||||
|
||||
if (LittleCores.empty()) {
|
||||
// If we only ended up with big cores then we need to move some to be little cores
|
||||
uint32_t LowestMIDR = ~0U;
|
||||
uint32_t LowestMIDRIdx = 0;
|
||||
// Walk all the big cores
|
||||
for (size_t i = 0; i < BigCores.size(); ++i) {
|
||||
uint8_t Implementer = BigCores[i]->Implementer;
|
||||
uint16_t Part = BigCores[i]->Part;
|
||||
|
||||
// Walk our list of CPUMIDRs to find the most little core
|
||||
for (size_t j = LowestMIDRIdx; j < CPUMIDRs.size(); ++j) {
|
||||
auto &MIDROption = CPUMIDRs[i];
|
||||
if ((MIDROption.Implementer == Implementer &&
|
||||
MIDROption.Part == Part) ||
|
||||
(MIDROption.Implementer == 0 &&
|
||||
MIDROption.Part == 0)) {
|
||||
|
||||
LowestMIDRIdx = j;
|
||||
LowestMIDR = MIDR;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Now we WILL have found a big core to demote to little status
|
||||
// Demote them
|
||||
std::erase_if(BigCores, [&LittleCores, LowestMIDR](auto *Entry) {
|
||||
// Demote by erase copy to little array
|
||||
uint8_t Implementer = LowestMIDR >> 24;
|
||||
uint16_t Part = (LowestMIDR >> 4) & 0xFFF;
|
||||
|
||||
if (Entry->Implementer == Implementer &&
|
||||
Entry->Part == Part) {
|
||||
// Add it to the BigCore list
|
||||
LittleCores.emplace_back(Entry);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
});
|
||||
}
|
||||
|
||||
if (BigCores.empty()) {
|
||||
// We never found a CPU core we understand
|
||||
// Grab the first core, consider it as little, move everything else to Big
|
||||
uint32_t LittleMIDR = PerCPUData[0].MIDR;
|
||||
// Now walk the little cores and move them to Big if they don't match
|
||||
std::erase_if(LittleCores, [&BigCores, LittleMIDR](auto *Entry) {
|
||||
// You're promoted now
|
||||
uint8_t Implementer = LittleMIDR >> 24;
|
||||
uint16_t Part = (LittleMIDR >> 4) & 0xFFF;
|
||||
|
||||
if (Entry->Implementer != Implementer ||
|
||||
Entry->Part != Part) {
|
||||
// Add it to the BigCore list
|
||||
BigCores.emplace_back(Entry);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
});
|
||||
}
|
||||
|
||||
// Now walk the per CPU data one more time and set if it is big or little
|
||||
for (auto &Data : PerCPUData) {
|
||||
uint8_t Implementer = Data.MIDR >> 24;
|
||||
uint16_t Part = (Data.MIDR >> 4) & 0xFFF;
|
||||
|
||||
bool FoundBig{};
|
||||
const CPUMIDR *MIDR{};
|
||||
for (auto Big : BigCores) {
|
||||
if (Big->Implementer == Implementer &&
|
||||
Big->Part == Part) {
|
||||
FoundBig = true;
|
||||
MIDR = Big;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!FoundBig) {
|
||||
for (auto Little : LittleCores) {
|
||||
if (Little->Implementer == Implementer &&
|
||||
Little->Part == Part) {
|
||||
MIDR = Little;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Data.IsBig = FoundBig;
|
||||
if (MIDR) {
|
||||
Data.ProductName = MIDR->ProductName ?: ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
else {
|
||||
Data.ProductName = ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If we aren't hybrid then just claim everything is big
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
|
||||
PerCPUData[i].IsBig = true;
|
||||
if (MIDROption) {
|
||||
PerCPUData[i].ProductName = MIDROption->ProductName ?: ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
else {
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
@@ -55,6 +362,24 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
__cpuid(0, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x7) {
|
||||
__cpuid(0x7, eax, ebx, ecx, edx);
|
||||
// Bit 15 of edx claims hybrid CPU
|
||||
Hybrid = (edx & (1U << 15)) != 0;
|
||||
}
|
||||
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
PerCPUData[i].IsBig = true;
|
||||
PerCPUData[i].ProductName = ProductNames::UNKNOWN;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
@@ -79,6 +404,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
@@ -108,7 +435,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(0 << 20) | // SSE4.2
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
@@ -305,12 +632,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(0 << 3) | // BMI1
|
||||
(1 << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(0 << 8) | // BMI2
|
||||
(1 << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
@@ -321,7 +648,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // RDSEED
|
||||
(0 << 19) | // ADCX and ADOX instructions
|
||||
(1 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
@@ -379,29 +706,29 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 6) | // Reserved
|
||||
(0 << 7) | // Reserved
|
||||
(0 << 8) | // AVX512_VP2INTERSECT
|
||||
(0 << 9) | // Reserved
|
||||
(0 << 9) | // SRBDS_CTRL (Special Register Buffer Data Sampling Mitigations)
|
||||
(0 << 10) | // VERW clears CPU buffers
|
||||
(0 << 11) | // Reserved
|
||||
(0 << 12) | // Reserved
|
||||
(0 << 13) | // Reserved
|
||||
(0 << 13) | // TSX Force Abort (TSX will force abort if attempted)
|
||||
(0 << 14) | // SERIALIZE instruction
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // Reserved
|
||||
((Hybrid ? 1U : 0U) << 15) | // Hybrid
|
||||
(0 << 16) | // TSXLDTRK (TSX Suspend load address tracking) - Allows untracked memory loads inside TSX region
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // Intel PCONFIG
|
||||
(0 << 19) | // Intel Architectural LBR
|
||||
(0 << 20) | // Intel CET
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 22) | // AMX-BF16 - Tile computation on bfloat16
|
||||
(0 << 23) | // AVX512_FP16 - FP16 AVX512 instructions
|
||||
(0 << 24) | // AMX-tile - If AMX is implemented
|
||||
(0 << 25) | // AMX-int8 - AMX on 8-bit integers
|
||||
(0 << 26) | // IBRS_IBPB - Speculation control
|
||||
(0 << 27) | // STIBP - Single Thread Indirect Branch Predictor, Part of IBC
|
||||
(0 << 28) | // L1D Flush
|
||||
(0 << 29) | // Arch capabilities
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
(0 << 29) | // Arch capabilities - Speculative side channel mitigations
|
||||
(0 << 30) | // Arch capabilities - MSR module specific
|
||||
(0 << 31); // SSBD - Speculative Store Bypass Disable
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -473,6 +800,18 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Hybrid) {
|
||||
uint32_t CPU = GetCPUID();
|
||||
auto &Data = PerCPUData[CPU];
|
||||
// 0x40 is a big CPU
|
||||
// 0x20 is a little CPU
|
||||
Res.eax |= (Data.IsBig ? 0x40 : 0x20) << 24;
|
||||
}
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Highest extended function implemented
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
@@ -560,7 +899,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(0 << 27) | // RDTSCP
|
||||
(1 << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(0 << 30) | // 3DNow! Extensions
|
||||
@@ -568,27 +907,44 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
constexpr char ProcessorBrand[48] = {
|
||||
constexpr char ProcessorBrand[32] = {
|
||||
GIT_DESCRIBE_STRING
|
||||
"\0"
|
||||
};
|
||||
|
||||
constexpr ssize_t DESCRIBE_STR_SIZE = std::char_traits<char>::length(GIT_DESCRIBE_STRING);
|
||||
static_assert(DESCRIBE_STR_SIZE < 32);
|
||||
|
||||
//Processor brand string
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[0], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
return Function_8000_0002h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[16], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
return Function_8000_0003h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
|
||||
return Function_8000_0004h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[32], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(16L, DESCRIBE_STR_SIZE));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(0L, DESCRIBE_STR_SIZE - 16));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
auto &Data = PerCPUData[CPU];
|
||||
memcpy(&Res, Data.ProductName, std::min(strlen(Data.ProductName), sizeof(FEXCore::CPUID::FunctionResults)));
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -681,7 +1037,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) {
|
||||
Res.ebx =
|
||||
(0 << 2) | // XSaveErPtr: Saving and restoring error pointers
|
||||
(0 << 1) | // IRPerf: Instructions retired count support
|
||||
(0 << 0); // CLZERO support
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
Res.ecx =
|
||||
@@ -812,25 +1168,25 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
CTX = ctx;
|
||||
using namespace std::placeholders;
|
||||
RegisterFunction(0, std::bind(&CPUIDEmu::Function_0h, this, _1));
|
||||
RegisterFunction(1, std::bind(&CPUIDEmu::Function_01h, this, _1));
|
||||
RegisterFunction(2, std::bind(&CPUIDEmu::Function_02h, this, _1));
|
||||
|
||||
RegisterFunction(0, &CPUIDEmu::Function_0h);
|
||||
RegisterFunction(1, &CPUIDEmu::Function_01h);
|
||||
RegisterFunction(2, &CPUIDEmu::Function_02h);
|
||||
// 3: Serial Number(previously), now reserved
|
||||
#ifndef CPUID_AMD
|
||||
// Deterministic cache parameters for each level
|
||||
RegisterFunction(0x4, std::bind(&CPUIDEmu::Function_04h, this, _1));
|
||||
RegisterFunction(0x4, &CPUIDEmu::Function_04h);
|
||||
#endif
|
||||
// 5: Monitor/mwait
|
||||
// Thermal and power management
|
||||
RegisterFunction(6, std::bind(&CPUIDEmu::Function_06h, this, _1));
|
||||
RegisterFunction(6, &CPUIDEmu::Function_06h);
|
||||
// Extended feature flags
|
||||
RegisterFunction(7, std::bind(&CPUIDEmu::Function_07h, this, _1));
|
||||
RegisterFunction(7, &CPUIDEmu::Function_07h);
|
||||
// 9: Direct Cache Access information
|
||||
// 0x0A: Architectural performance monitoring
|
||||
// 0x0B: Extended topology enumeration
|
||||
// 0x0D: Processor extended state enumeration
|
||||
RegisterFunction(0x0D, std::bind(&CPUIDEmu::Function_0Dh, this, _1));
|
||||
RegisterFunction(0x0D, &CPUIDEmu::Function_0Dh);
|
||||
// 0x0F: Intel RDT monitoring
|
||||
// 0x10: Intel RDT allocation enumeration
|
||||
// 0x12: Intel SGX capability enumeration
|
||||
@@ -839,38 +1195,42 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
#ifndef CPUID_AMD
|
||||
// Timestamp counter information
|
||||
// Doesn't exist on AMD hardware
|
||||
RegisterFunction(0x15, std::bind(&CPUIDEmu::Function_15h, this, _1));
|
||||
RegisterFunction(0x15, &CPUIDEmu::Function_15h);
|
||||
#endif
|
||||
// 0x16: Processor frequency information
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
#ifndef CPUID_AMD
|
||||
RegisterFunction(0x1A, &CPUIDEmu::Function_1Ah);
|
||||
#endif
|
||||
// Largest extended function number
|
||||
RegisterFunction(0x8000'0000, std::bind(&CPUIDEmu::Function_8000_0000h, this, _1));
|
||||
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
|
||||
// Processor vendor
|
||||
RegisterFunction(0x8000'0001, std::bind(&CPUIDEmu::Function_8000_0001h, this, _1));
|
||||
RegisterFunction(0x8000'0001, &CPUIDEmu::Function_8000_0001h);
|
||||
// Processor brand string
|
||||
RegisterFunction(0x8000'0002, std::bind(&CPUIDEmu::Function_8000_0002h, this, _1));
|
||||
RegisterFunction(0x8000'0002, &CPUIDEmu::Function_8000_0002h);
|
||||
// Processor brand string continued
|
||||
RegisterFunction(0x8000'0003, std::bind(&CPUIDEmu::Function_8000_0003h, this, _1));
|
||||
RegisterFunction(0x8000'0003, &CPUIDEmu::Function_8000_0003h);
|
||||
// Processor brand string continued
|
||||
RegisterFunction(0x8000'0004, std::bind(&CPUIDEmu::Function_8000_0004h, this, _1));
|
||||
RegisterFunction(0x8000'0004, &CPUIDEmu::Function_8000_0004h);
|
||||
// 0x8000'0005: L1 Cache and TLB identifiers
|
||||
#ifdef CPUID_AMD
|
||||
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_8000_0005h, this, _1));
|
||||
RegisterFunction(0x8000'0005, &CPUIDEmu::Function_8000_0005h);
|
||||
#else
|
||||
// This is full reserved on Intel platforms
|
||||
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_Reserved, this, _1));
|
||||
RegisterFunction(0x8000'0005, &CPUIDEmu::Function_Reserved);
|
||||
#endif
|
||||
// 0x8000'0006: L2 Cache identifiers
|
||||
RegisterFunction(0x8000'0006, std::bind(&CPUIDEmu::Function_8000_0006h, this, _1));
|
||||
RegisterFunction(0x8000'0006, &CPUIDEmu::Function_8000_0006h);
|
||||
// Advanced power management information
|
||||
RegisterFunction(0x8000'0007, std::bind(&CPUIDEmu::Function_8000_0007h, this, _1));
|
||||
RegisterFunction(0x8000'0007, &CPUIDEmu::Function_8000_0007h);
|
||||
// Virtual and physical address sizes
|
||||
RegisterFunction(0x8000'0008, std::bind(&CPUIDEmu::Function_8000_0008h, this, _1));
|
||||
RegisterFunction(0x8000'0008, &CPUIDEmu::Function_8000_0008h);
|
||||
|
||||
// 0x8000'000A: SVM Revision
|
||||
// TLB 1GB page identifiers
|
||||
RegisterFunction(0x8000'0019, std::bind(&CPUIDEmu::Function_8000_0019h, this, _1));
|
||||
RegisterFunction(0x8000'0019, &CPUIDEmu::Function_8000_0019h);
|
||||
|
||||
// 0x8000'001A: Performance optimization identifiers
|
||||
// 0x8000'001B: Instruction based sampling identifiers
|
||||
@@ -878,10 +1238,13 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
// 0x8000'001D: Cache properties
|
||||
#ifdef CPUID_AMD
|
||||
// Deterministic cache parameters for each level
|
||||
RegisterFunction(0x8000'001D, std::bind(&CPUIDEmu::Function_8000_001Dh, this, _1));
|
||||
RegisterFunction(0x8000'001D, &CPUIDEmu::Function_8000_001Dh);
|
||||
#endif
|
||||
// 0x8000'001E: Extended APIC ID
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
}
|
||||
}
|
||||
|
||||
+39
-6
@@ -1,10 +1,11 @@
|
||||
#pragma once
|
||||
#include <functional>
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -22,27 +23,50 @@ private:
|
||||
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
|
||||
|
||||
public:
|
||||
// X86 cacheline size effectively has to be hardcoded to 64
|
||||
// if we report anything differently then applications are likely to break
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
|
||||
void Init(FEXCore::Context::Context *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
|
||||
auto Handler = FunctionHandlers.find(Function);
|
||||
const auto Handler = FunctionHandlers.find(Function);
|
||||
|
||||
if (Handler == FunctionHandlers.end()) {
|
||||
return Function_Reserved(Leaf);
|
||||
}
|
||||
|
||||
return Handler->second(Leaf);
|
||||
return (this->*Handler->second)(Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
if (Function == 0x8000'0002U)
|
||||
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
|
||||
else if (Function == 0x8000'0003U)
|
||||
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
|
||||
else
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
|
||||
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults(uint32_t Leaf)>;
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
|
||||
FunctionHandlers[Function] = Handler;
|
||||
FunctionHandlers.insert_or_assign(Function, Handler);
|
||||
}
|
||||
|
||||
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
uint32_t MIDR{};
|
||||
#endif
|
||||
bool IsBig{};
|
||||
};
|
||||
std::vector<CPUData> PerCPUData{};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
@@ -53,11 +77,17 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU);
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
|
||||
@@ -66,5 +96,8 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
|
||||
};
|
||||
}
|
||||
+41
-44
@@ -1,8 +1,21 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
#include "FEXCore/HLE/Linux/ThreadManagement.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <memory>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
|
||||
namespace FEXCore {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
@@ -22,7 +35,9 @@ namespace FEXCore {
|
||||
CTX->InitializeCompiler(CompileThreadData.get(), true);
|
||||
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CompileService::Initialize() {
|
||||
@@ -45,23 +60,15 @@ namespace FEXCore {
|
||||
// Grab the work queue and clear it
|
||||
// We don't need to grab the queue mutex since this thread will no longer receive any work events
|
||||
// Threads are bounded 1:1
|
||||
while (WorkQueue.size()) {
|
||||
WorkItem *Item = WorkQueue.front();
|
||||
while (!WorkQueue.empty()) {
|
||||
WorkQueue.pop();
|
||||
delete Item;
|
||||
}
|
||||
|
||||
// Go through the garbage collection array and clear it
|
||||
// It's safe to clear things that aren't marked safe since we are clearing cache
|
||||
if (GCArray.size()) {
|
||||
// Clean up our GC array
|
||||
for (auto it = GCArray.begin(); it != GCArray.end();) {
|
||||
delete *it;
|
||||
it = GCArray.erase(it);
|
||||
}
|
||||
}
|
||||
GCArray.clear();
|
||||
|
||||
LOGMAN_THROW_A(CompileThreadData->LocalIRCache.size() == 0, "Compile service must never have LocalIRCache");
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LocalIRCache.empty(), "Compile service must never have LocalIRCache");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
@@ -72,28 +79,26 @@ namespace FEXCore {
|
||||
SelectedThread->CPUBackend->ClearCache();
|
||||
}
|
||||
|
||||
|
||||
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
|
||||
// Tell the worker thread to compile code for us
|
||||
WorkItem *Item = new WorkItem{};
|
||||
Item->RIP = RIP;
|
||||
WorkItem* ResultItem = nullptr;
|
||||
|
||||
{
|
||||
// Tell the worker thread to compile code for us
|
||||
auto Item = std::make_unique<WorkItem>();
|
||||
Item->RIP = RIP;
|
||||
|
||||
// Fill the threads work queue
|
||||
std::scoped_lock<std::mutex> lk(QueueMutex);
|
||||
WorkQueue.emplace(Item);
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
ResultItem = WorkQueue.emplace(std::move(Item)).get();
|
||||
}
|
||||
|
||||
// Notify the thread that it has more work
|
||||
StartWork.NotifyAll();
|
||||
|
||||
return Item;
|
||||
return ResultItem;
|
||||
}
|
||||
|
||||
void CompileService::ExecutionThread() {
|
||||
// Ignore signals coming from the guest
|
||||
CTX->SignalDelegation->MaskThreadSignals();
|
||||
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16]{};
|
||||
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
|
||||
@@ -105,18 +110,18 @@ namespace FEXCore {
|
||||
if (ShuttingDown.load()) {
|
||||
break;
|
||||
}
|
||||
std::scoped_lock<std::mutex> lk(CompileMutex);
|
||||
|
||||
std::scoped_lock lk(CompileMutex);
|
||||
size_t WorkItems{};
|
||||
|
||||
do {
|
||||
// Grab a work item
|
||||
WorkItem *Item{};
|
||||
std::unique_ptr<WorkItem> Item{};
|
||||
{
|
||||
std::scoped_lock<std::mutex> lk(QueueMutex);
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
WorkItems = WorkQueue.size();
|
||||
if (WorkItems) {
|
||||
Item = WorkQueue.front();
|
||||
if (WorkItems != 0) {
|
||||
Item = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
}
|
||||
@@ -124,7 +129,7 @@ namespace FEXCore {
|
||||
// If we had a work item then work on it
|
||||
if (Item) {
|
||||
// Make sure it's not in lookup cache by accident
|
||||
LOGMAN_THROW_A(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
@@ -132,11 +137,11 @@ namespace FEXCore {
|
||||
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
LOGMAN_THROW_A(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
LOGMAN_THROW_A_FMT(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
if (!CodePtr) {
|
||||
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
|
||||
ERROR_AND_DIE("Couldn't compile code for thread at RIP: 0x%lx", Item->RIP);
|
||||
ERROR_AND_DIE_FMT("Couldn't compile code for thread at RIP: 0x{:x}", Item->RIP);
|
||||
}
|
||||
|
||||
Item->CodePtr = CodePtr;
|
||||
@@ -146,23 +151,15 @@ namespace FEXCore {
|
||||
Item->StartAddr = StartAddr;
|
||||
Item->Length = Length;
|
||||
|
||||
GCArray.emplace_back(Item);
|
||||
Item->ServiceWorkDone.NotifyAll();
|
||||
auto& GCItem = GCArray.emplace_back(std::move(Item));
|
||||
GCItem->ServiceWorkDone.NotifyAll();
|
||||
}
|
||||
} while (WorkItems != 0);
|
||||
|
||||
if (GCArray.size()) {
|
||||
// Clean up our GC array
|
||||
for (auto it = GCArray.begin(); it != GCArray.end();) {
|
||||
if ((*it)->SafeToClear) {
|
||||
delete *it;
|
||||
it = GCArray.erase(it);
|
||||
}
|
||||
else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Clean up any safe entries in our GC array if we have any.
|
||||
std::erase_if(GCArray, [](const auto& Entry) {
|
||||
return Entry->SafeToClear.load(std::memory_order_relaxed);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
+11
-9
@@ -1,23 +1,22 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
#include <thread>
|
||||
#include <unordered_map>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <stdint.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
namespace Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
};
|
||||
class CompileService final {
|
||||
@@ -49,6 +48,9 @@ class CompileService final {
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address) const {
|
||||
return CompileThreadData->CPUBackend->IsAddressInJITCode(Address, false, false);
|
||||
}
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ParentThread;
|
||||
@@ -58,8 +60,8 @@ class CompileService final {
|
||||
|
||||
std::mutex QueueMutex{};
|
||||
std::mutex CompileMutex{};
|
||||
std::queue<WorkItem*> WorkQueue{};
|
||||
std::vector<WorkItem*> GCArray{};
|
||||
std::queue<std::unique_ptr<WorkItem>> WorkQueue{};
|
||||
std::vector<std::unique_ptr<WorkItem>> GCArray{};
|
||||
Event StartWork{};
|
||||
std::atomic_bool ShuttingDown{false};
|
||||
};
|
||||
|
||||
+234
-481
@@ -7,42 +7,71 @@ desc: Glues Frontend, OpDispatcher and IR Opts & Compilation, LookupCache, Dispa
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Common/Paths.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/DebugData.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterCore.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "FEXCore/Utils/Allocator.h"
|
||||
|
||||
#include <xxhash.h>
|
||||
#include <fstream>
|
||||
#include <unistd.h>
|
||||
#include <filesystem>
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <functional>
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <set>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/mman.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
#include <sys/syscall.h>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(FEXCore::Context::Context *CTX) {
|
||||
@@ -113,63 +142,11 @@ std::string_view const& GetGRegName(unsigned Reg) {
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void Context::AOTIRCaptureCacheWriteoutQueue_Flush() {
|
||||
{
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
AOTIRCaptureCacheWriteoutLock.unlock();
|
||||
|
||||
fn();
|
||||
if (MaybeEmpty) {
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A("Must never get here");
|
||||
}
|
||||
|
||||
void Context::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
std::unique_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
AOTIRCaptureCacheWriteoutQueue.push(fn);
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() > 10000) {
|
||||
Flush = true;
|
||||
}
|
||||
}
|
||||
|
||||
bool test_val = false;
|
||||
if (Flush && AOTIRCaptureCacheWriteoutFlusing.compare_exchange_strong(test_val, true)) {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
}
|
||||
}
|
||||
|
||||
Context::Context() {
|
||||
Context::Context()
|
||||
: IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
else {
|
||||
StopGdbServer();
|
||||
}
|
||||
}
|
||||
|
||||
Context::~Context() {
|
||||
@@ -189,17 +166,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
Threads.clear();
|
||||
}
|
||||
|
||||
for (auto &Mod: AOTIRCache) {
|
||||
FEXCore::Allocator::munmap(Mod.second.mapping, Mod.second.size);
|
||||
}
|
||||
}
|
||||
|
||||
bool Context::InitCore(FEXCore::CodeLoader *Loader) {
|
||||
ThunkHandler.reset(FEXCore::ThunkHandler::Create());
|
||||
|
||||
LocalLoader = Loader;
|
||||
using namespace FEXCore::Core;
|
||||
static FEXCore::Core::CPUState CreateDefaultCPUState() {
|
||||
FEXCore::Core::CPUState NewThreadState{};
|
||||
|
||||
// Initialize default CPU state
|
||||
@@ -217,7 +186,23 @@ namespace FEXCore::Context {
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
NewThreadState.FTW = 0xFFFF;
|
||||
return NewThreadState;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* Context::InitCore(FEXCore::CodeLoader *Loader) {
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
else {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
ThunkHandler.reset(FEXCore::ThunkHandler::Create());
|
||||
|
||||
LocalLoader = Loader;
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0);
|
||||
|
||||
// We are the parent thread
|
||||
@@ -228,8 +213,7 @@ namespace FEXCore::Context {
|
||||
Thread->CurrentFrame->State.rip = StartingRIP = Loader->DefaultRIP();
|
||||
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
return true;
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void Context::StartGdbServer() {
|
||||
@@ -243,8 +227,7 @@ namespace FEXCore::Context {
|
||||
DebugServer.reset();
|
||||
}
|
||||
|
||||
void Context::HandleCallback(uint64_t RIP) {
|
||||
auto Thread = Core::ThreadData.Thread;
|
||||
void Context::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
Thread->CPUBackend->CallbackPtr(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
@@ -291,7 +274,7 @@ namespace FEXCore::Context {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Pause);
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
// Only attempt to stop this thread if it is running
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -310,7 +293,6 @@ namespace FEXCore::Context {
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Return);
|
||||
Thread->RunningEvents.WaitingToStart.store(true);
|
||||
}
|
||||
|
||||
for (auto &Thread : Threads) {
|
||||
@@ -349,13 +331,13 @@ namespace FEXCore::Context {
|
||||
this->Config.MaxInstPerBlock = 1;
|
||||
Run();
|
||||
WaitForThreadsToRun();
|
||||
WaitForIdleWithTimeout();
|
||||
WaitForIdle();
|
||||
this->Config.RunningMode = PreviousRunningMode;
|
||||
this->Config.MaxInstPerBlock = PreviousMaxIntPerBlock;
|
||||
}
|
||||
|
||||
void Context::Stop(bool IgnoreCurrentThread) {
|
||||
pid_t tid = gettid();
|
||||
pid_t tid = FHU::Syscalls::gettid();
|
||||
FEXCore::Core::InternalThreadState* CurrentThread{};
|
||||
|
||||
// Tell all the threads that they should stop
|
||||
@@ -376,6 +358,13 @@ namespace FEXCore::Context {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
StopThread(Thread);
|
||||
}
|
||||
|
||||
// If the thread is waiting to start but immediately killed then there can be a hang
|
||||
// This occurs in the case of gdb attach with immediate kill
|
||||
if (Thread->RunningEvents.WaitingToStart.load()) {
|
||||
Thread->RunningEvents.EarlyExit = true;
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -388,20 +377,25 @@ namespace FEXCore::Context {
|
||||
void Context::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->RunningEvents.Running.exchange(false)) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Stop);
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
|
||||
void Context::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
Thread->SignalReason.store(Event);
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason Context::RunUntilExit() {
|
||||
if(!StartPaused)
|
||||
Run();
|
||||
if(!StartPaused) {
|
||||
// We will only have one thread at this point, but just in case run notify everything
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
|
||||
ExecutionThread(ParentThread);
|
||||
while(true) {
|
||||
@@ -425,12 +419,17 @@ namespace FEXCore::Context {
|
||||
auto IRHandler = [Thread](uint64_t Addr, IR::IREmitter *IR) -> void {
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IR);
|
||||
Core::LocalIREntry Entry = {Addr, 0ULL, decltype(Entry.IR)(IR->CreateIRCopy()), decltype(Entry.RAData)(Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->PullAllocationData() : nullptr), decltype(Entry.DebugData)(new Core::DebugData())};
|
||||
Core::LocalIREntry Entry = {Addr, 0ULL,
|
||||
decltype(Entry.IR)(IR->CreateIRCopy()),
|
||||
decltype(Entry.RAData)(Thread->PassManager->HasPass("RA")
|
||||
? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData()
|
||||
: nullptr),
|
||||
decltype(Entry.DebugData)(new Core::DebugData())
|
||||
};
|
||||
Thread->LocalIRCache.insert({Addr, std::move(Entry)});
|
||||
};
|
||||
|
||||
LocalLoader->AddIR(IRHandler);
|
||||
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
@@ -456,6 +455,14 @@ namespace FEXCore::Context {
|
||||
Thread->ThreadWaiting.Wait();
|
||||
}
|
||||
|
||||
void Context::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
|
||||
void Context::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Tell the thread to start executing
|
||||
Thread->StartRunning.NotifyAll();
|
||||
@@ -471,8 +478,10 @@ namespace FEXCore::Context {
|
||||
Stop(false /* Ignore current thread */);
|
||||
});
|
||||
|
||||
State->CTX = this;
|
||||
|
||||
#if _M_ARM_64
|
||||
bool DoSRA = true;
|
||||
bool DoSRA = State->CTX->Config.StaticRegisterAllocation;
|
||||
#else
|
||||
bool DoSRA = false;
|
||||
#endif
|
||||
@@ -482,13 +491,13 @@ namespace FEXCore::Context {
|
||||
|
||||
State->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
State->CTX = this;
|
||||
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread);
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
State->PassManager->InsertRegisterAllocationPass(DoSRA);
|
||||
|
||||
@@ -497,14 +506,16 @@ namespace FEXCore::Context {
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
State->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, State, CompileThread);
|
||||
#else
|
||||
ERROR_AND_DIE("FEXCore has been compiled without a viable JIT core");
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
#endif
|
||||
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM:
|
||||
State->CPUBackend = CustomCPUFactory(this, State);
|
||||
break;
|
||||
default: ERROR_AND_DIE("Unknown core configuration");
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown core configuration");
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -537,7 +548,7 @@ namespace FEXCore::Context {
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
|
||||
auto It = std::find(Threads.begin(), Threads.end(), Thread);
|
||||
LOGMAN_THROW_A(It != Threads.end(), "Thread wasn't in Threads");
|
||||
LOGMAN_THROW_A_FMT(It != Threads.end(), "Thread wasn't in Threads");
|
||||
|
||||
Threads.erase(It);
|
||||
}
|
||||
@@ -611,6 +622,58 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
const auto DumpIRStr = Thread->CTX->Config.DumpIR();
|
||||
|
||||
if (DumpIRStr =="stderr") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
const auto fileName = fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
f = fopen(fileName.c_str(), "w");
|
||||
CloseAfter = true;
|
||||
}
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fmt::print(f,"IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static void ValidateIR(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction();
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
auto reparsed = IR::Parse(&out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
std::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
LogMan::Msg::IFmt("one:\n {}", out.str());
|
||||
LogMan::Msg::IFmt("two:\n {}", out2.str());
|
||||
LOGMAN_MSG_A_FMT("Parsed IR doesn't match\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
@@ -620,9 +683,7 @@ namespace FEXCore::Context {
|
||||
uint64_t TotalInstructions {0};
|
||||
uint64_t TotalInstructionsLength {0};
|
||||
|
||||
if (!Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP)) {
|
||||
return {};
|
||||
}
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP);
|
||||
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
|
||||
@@ -635,7 +696,6 @@ namespace FEXCore::Context {
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
|
||||
// Reset any block-specific state
|
||||
@@ -643,11 +703,6 @@ namespace FEXCore::Context {
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
if (Block.HasInvalidInstruction) {
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
break;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr};
|
||||
FEXCore::X86Tables::DecodedInst const* DecodedInfo {nullptr};
|
||||
@@ -677,7 +732,7 @@ namespace FEXCore::Context {
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
}
|
||||
|
||||
if (TableInfo->OpcodeDispatcher) {
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->HandledLock = false;
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
@@ -688,7 +743,7 @@ namespace FEXCore::Context {
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::E("Missing LOCK HANDLER at 0x%lx{'%s'}", Block.Entry + BlockInstructionsLength, TableInfo->Name);
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
@@ -696,8 +751,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Missing OpDispatcher at 0x%lx{'%s'}", Block.Entry + BlockInstructionsLength, TableInfo->Name);
|
||||
HadDispatchError = true;
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
}
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
@@ -724,77 +780,35 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
auto IRDumper = [Thread, GuestRIP](IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
|
||||
if (Thread->CTX->Config.DumpIR() =="stderr") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (Thread->CTX->Config.DumpIR() =="stdout") {
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
std::stringstream fileName;
|
||||
fileName << Thread->CTX->Config.DumpIR() << "/" << std::hex << GuestRIP << (RA ? "-post.ir" : "-pre.ir");
|
||||
|
||||
f = fopen(fileName.str().c_str(), "w");
|
||||
CloseAfter = true;
|
||||
// Debug
|
||||
{
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread, GuestRIP, nullptr);
|
||||
}
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fprintf(f,"IR-%s 0x%lx:\n%s\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str().c_str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(nullptr);
|
||||
}
|
||||
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction();
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
auto reparsed = IR::Parse(&out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A("Failed to parse ir\n");
|
||||
} else {
|
||||
std::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
LogMan::Msg::I("one:\n %s", out.str().c_str());
|
||||
LogMan::Msg::I("two:\n %s", out2.str().c_str());
|
||||
LOGMAN_MSG_A("Parsed ir doesn't match\n");
|
||||
}
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
ValidateIR(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(Thread->OpDispatcher.get());
|
||||
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->GetAllocationData() : nullptr);
|
||||
// Debug
|
||||
{
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread, GuestRIP, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
LogMan::Msg::IFmt("IR 0x{:x}:\n{}\n@@@@@\n", GuestRIP, out.str());
|
||||
}
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->GetAllocationData() : nullptr);
|
||||
LogMan::Msg::I("IR 0x%lx:\n%s\n@@@@@\n", GuestRIP, out.str().c_str());
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->PullAllocationData() : nullptr;
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
auto IRList = Thread->OpDispatcher->CreateIRCopy();
|
||||
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
@@ -809,63 +823,6 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
uintptr_t This = (uintptr_t)this;
|
||||
|
||||
return (AOTIRInlineEntry*)(This + DataBase + DataOffset);
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
ssize_t l = 0;
|
||||
ssize_t r = Count - 1;
|
||||
|
||||
while (l <= r) {
|
||||
size_t m = l + (r - l) / 2;
|
||||
|
||||
if (Entries[m].GuestStart == GuestStart)
|
||||
return GetInlineEntry(Entries[m].DataOffset);
|
||||
else if (Entries[m].GuestStart < GuestStart)
|
||||
l = m + 1;
|
||||
else
|
||||
r = m - 1;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
IR::RegisterAllocationData *AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData *)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView *AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView *)&InlineData[Offset];
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->tellp());
|
||||
|
||||
if (Inserted.second) {
|
||||
//GuestHash
|
||||
Stream->write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
//GuestLength
|
||||
Stream->write((const char*)&Length, sizeof(Length));
|
||||
|
||||
// RAData (inline)
|
||||
// In file, IsShared is always set
|
||||
auto Shared = RAData->IsShared;
|
||||
RAData->IsShared = true;
|
||||
Stream->write((const char*)RAData, RAData->Size(RAData->MapCount));
|
||||
RAData->IsShared = Shared;
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
}
|
||||
}
|
||||
|
||||
Context::CompileCodeResult Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
@@ -889,55 +846,17 @@ namespace FEXCore::Context {
|
||||
GeneratedIR = false;
|
||||
}
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (!file->second.ContainsCode) {
|
||||
file->second.ContainsCode = true;
|
||||
FilesWithCode[file->second.fileid] = file->second.filename;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (IRList == nullptr && Config.AOTIRLoad) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
auto Mod = (AOTIRInlineIndex*)file->second.CachedFileEntry;
|
||||
|
||||
if (Mod == nullptr) {
|
||||
file->second.CachedFileEntry = Mod = AOTIRCache[file->second.fileid].Array;
|
||||
}
|
||||
|
||||
if (Mod != nullptr)
|
||||
{
|
||||
auto AOTEntry = Mod->Find(GuestRIP - file->second.Start + file->second.Offset);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
auto MappedStart = GuestRIP;
|
||||
auto hash = XXH3_64bits((void*)MappedStart, AOTEntry->GuestLength);
|
||||
if (hash == AOTEntry->GuestHash) {
|
||||
IRList = AOTEntry->GetIRData();
|
||||
//LogMan::Msg::D("using %s + %lx -> %lx\n", file->second.fileid.c_str(), AOTEntry->first, GuestRIP);
|
||||
|
||||
|
||||
RAData = AOTEntry->GetRAData();;
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = MappedStart;
|
||||
Length = AOTEntry->GuestLength;
|
||||
|
||||
GeneratedIR = true;
|
||||
} else {
|
||||
LogMan::Msg::I("AOTIR: hash check failed %lx\n", MappedStart);
|
||||
}
|
||||
} else {
|
||||
//LogMan::Msg::I("AOTIR: Failed to find %lx, %lx, %s\n", GuestRIP, GuestRIP - file->second.Start + file->second.Offset, file->second.fileid.c_str());
|
||||
}
|
||||
}
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = RACopy;
|
||||
DebugData = DebugDataCopy;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
GeneratedIR = _GeneratedIR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -978,120 +897,14 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
static bool readAll(int fd, void *data, size_t size) {
|
||||
int rv = read(fd, data, size);
|
||||
|
||||
if (rv != size)
|
||||
return false;
|
||||
else
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Context::LoadAOTIRCache(int streamfd) {
|
||||
uint64_t tag;
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != 0xDEADBEEFC0D30004)
|
||||
return false;
|
||||
|
||||
std::string Module;
|
||||
uint64_t ModSize;
|
||||
uint64_t IndexSize;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&ModSize, sizeof(ModSize)))
|
||||
return false;
|
||||
|
||||
Module.resize(ModSize);
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize, SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size()))
|
||||
return false;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize)))
|
||||
return false;
|
||||
|
||||
struct stat fileinfo;
|
||||
if (fstat(streamfd, &fileinfo) < 0)
|
||||
return false;
|
||||
size_t Size = (fileinfo.st_size + 4095) & ~4095;
|
||||
|
||||
size_t IndexOffset = fileinfo.st_size - IndexSize -sizeof(ModSize) - ModSize - sizeof(IndexSize);
|
||||
|
||||
void *FilePtr = FEXCore::Allocator::mmap(nullptr, Size, PROT_READ, MAP_SHARED, streamfd, 0);
|
||||
|
||||
if (FilePtr == MAP_FAILED)
|
||||
return false;
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
AOTIRCache.insert({Module, {Array, FilePtr, Size}});
|
||||
|
||||
LogMan::Msg::D("AOTIR: Module %s has %ld functions", Module.c_str(), Array->Count);
|
||||
|
||||
return true;
|
||||
|
||||
}
|
||||
|
||||
void Context::WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &File: FilesWithCode) {
|
||||
Writer(File.first, File.second);
|
||||
}
|
||||
}
|
||||
|
||||
void Context::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
for (auto& [String, Entry] : AOTIRCaptureCache) {
|
||||
if (!Entry.Stream) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto ModSize = String.size();
|
||||
auto &stream = Entry.Stream;
|
||||
|
||||
// pad to 32 bytes
|
||||
constexpr char Zero = 0;
|
||||
while(stream->tellp() & 31)
|
||||
stream->write(&Zero, 1);
|
||||
|
||||
// AOTIRInlineIndex
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->tellp();
|
||||
|
||||
stream->write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->write((const char*)&DataBase, sizeof(DataBase));
|
||||
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
//AOTIRInlineIndexEntry
|
||||
|
||||
// GuestStart
|
||||
stream->write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
|
||||
// DataOffset
|
||||
stream->write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
}
|
||||
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
stream->write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->write(String.c_str(), ModSize);
|
||||
stream->write((const char*)&ModSize, sizeof(ModSize));
|
||||
}
|
||||
}
|
||||
|
||||
void Context::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto NewBlock = CompileBlock(Frame, GuestRIP);
|
||||
|
||||
if (NewBlock == 0) {
|
||||
LogMan::Msg::E("CompileBlockJit: Failed to compile code %lX - aborting process", GuestRIP);
|
||||
abort();
|
||||
LogMan::Msg::EFmt("CompileBlockJit: Failed to compile code {:X} - aborting process", GuestRIP);
|
||||
// Return similar behaviour of SIGILL abort
|
||||
Frame->Thread->StatusCode = 128 + SIGILL;
|
||||
Stop(false /* Ignore current thread */);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1119,7 +932,7 @@ namespace FEXCore::Context {
|
||||
Thread->CompileService->Initialize();
|
||||
}
|
||||
|
||||
auto WorkItem = Thread->CompileService->CompileCode(GuestRIP);
|
||||
auto* WorkItem = Thread->CompileService->CompileCode(GuestRIP);
|
||||
WorkItem->ServiceWorkDone.Wait();
|
||||
// Return here with the data in place
|
||||
CodePtr = WorkItem->CodePtr;
|
||||
@@ -1154,63 +967,31 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
// The core managed to compile the code.
|
||||
#if ENABLE_JITSYMBOLS
|
||||
if (DebugData) {
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
|
||||
if (Config.BlockJITNaming()) {
|
||||
if (DebugData) {
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
} else {
|
||||
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
}
|
||||
} else {
|
||||
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if ((Config.AOTIRCapture() || Config.AOTIRGenerate()) && RAData) {
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto file = AddrToFile.lower_bound(StartAddr);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (file->second.Start <= StartAddr && (file->second.Start + file->second.Len) >= (StartAddr + Length)) {
|
||||
auto LocalRIP = GuestRIP - file->second.Start + file->second.Offset;
|
||||
auto LocalStartAddr = StartAddr - file->second.Start + file->second.Offset;
|
||||
auto fileid = file->second.fileid;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRList, RAData, fileid]() {
|
||||
auto *AotFile = &AOTIRCaptureCache[fileid];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(fileid);
|
||||
uint64_t tag = 0xDEADBEEFC0D30004;
|
||||
AotFile->Stream->write((char*)&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRList, RAData);
|
||||
delete IRList;
|
||||
FEXCore::Allocator::free(RAData);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
Thread->CPUBackend->ClearCache();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
}
|
||||
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
if (IRCaptureCache.PostCompileCode(
|
||||
Thread,
|
||||
CodePtr,
|
||||
GuestRIP,
|
||||
StartAddr,
|
||||
Length,
|
||||
RAData,
|
||||
IRList,
|
||||
DebugData,
|
||||
GeneratedIR,
|
||||
DecrementRefCount)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
if (DecrementRefCount)
|
||||
@@ -1226,11 +1007,7 @@ namespace FEXCore::Context {
|
||||
Core::ThreadData.Thread = Thread;
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
// Let's do some initial bookkeeping here
|
||||
Thread->ThreadManager.TID = ::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
InitializeThreadTLSData(Thread);
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
@@ -1242,14 +1019,17 @@ namespace FEXCore::Context {
|
||||
Thread->StartRunning.Wait();
|
||||
}
|
||||
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE;
|
||||
if (!Thread->RunningEvents.EarlyExit.load()) {
|
||||
Thread->RunningEvents.WaitingToStart = false;
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE;
|
||||
|
||||
Thread->CPUBackend->ExecuteDispatch(Thread->CurrentFrame);
|
||||
Thread->RunningEvents.Running = true;
|
||||
|
||||
Thread->RunningEvents.WaitingToStart = false;
|
||||
Thread->RunningEvents.Running = false;
|
||||
Thread->CPUBackend->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
Thread->RunningEvents.Running = false;
|
||||
}
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
// XXX: This doesn't make sense when the parent thread doesn't outlive its children
|
||||
@@ -1341,38 +1121,11 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// TODO: Support overlapping maps and region splitting
|
||||
auto base_filename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (base_filename.size()) {
|
||||
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
|
||||
|
||||
auto fileid = base_filename + "-" + std::to_string(filename_hash) + "-";
|
||||
|
||||
// append optimization flags to the fileid
|
||||
fileid += (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? "S" : "s";
|
||||
fileid += Config.TSOEnabled ? "T" : "t";
|
||||
fileid += Config.ABILocalFlags ? "L" : "l";
|
||||
fileid += Config.ABINoPF ? "p" : "P";
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
AddrToFile.insert({ Base, { Base, Size, Offset, fileid, filename, nullptr, false} });
|
||||
|
||||
if (Config.AOTIRLoad && !AOTIRCache.contains(fileid) && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
if (streamfd != -1) {
|
||||
LoadAOTIRCache(streamfd);
|
||||
close(streamfd);
|
||||
}
|
||||
}
|
||||
}
|
||||
IRCaptureCache.AddNamedRegion(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
void Context::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
// TODO: Support partial removing
|
||||
AddrToFile.erase(Base);
|
||||
IRCaptureCache.RemoveNamedRegion(Base, Size);
|
||||
}
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
|
||||
@@ -2,17 +2,33 @@
|
||||
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "code-buffer-vixl.h"
|
||||
#include "platform-vixl.h"
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -23,12 +39,11 @@ static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(MAX_DISPATCHER_CODE_SIZE) {
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
SetAllowAssembler(true);
|
||||
|
||||
auto Buffer = GetBuffer();
|
||||
DispatchPtr = Buffer->GetOffsetAddress<CPUBackend::AsmDispatch>(GetCursorOffset());
|
||||
DispatchPtr = GetCursorAddress<CPUBackend::AsmDispatch>();
|
||||
|
||||
// while (true) {
|
||||
// Ptr = FindBlock(RIP)
|
||||
@@ -61,7 +76,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
add(x0, sp, 0);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled) {
|
||||
FillStaticRegs();
|
||||
@@ -180,11 +195,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
|
||||
{
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadStopHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PopCalleeSavedRegisters();
|
||||
|
||||
@@ -193,11 +208,31 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
{
|
||||
ExitFunctionLinkerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(x0, ~0ULL);
|
||||
stp(x0, x0, MemOperand(sp, -16, PreIndex));
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
add(x2, sp, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
|
||||
ldr(x0, &l_ExitFunctionLinkThis);
|
||||
mov(x1, STATE);
|
||||
mov(x2, lr);
|
||||
@@ -205,6 +240,24 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
ldr(x3, &l_ExitFunctionLink);
|
||||
blr(x3);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(x4, x0);
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
LoadConstant(x2, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(sp, sp, 16);
|
||||
|
||||
mov(x0, x4);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
FillStaticRegs();
|
||||
br(x0);
|
||||
@@ -214,16 +267,53 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
bind(&NoBlock);
|
||||
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(x0, ~0ULL);
|
||||
stp(x0, x2, MemOperand(sp, -16, PreIndex));
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
add(x2, sp, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(x2, MemOperand(sp, 8, Offset));
|
||||
}
|
||||
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x3, &l_CompileBlock);
|
||||
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
// X2 contains our guest RIP
|
||||
blr(x3); // { CTX, Frame, RIP}
|
||||
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
LoadConstant(x2, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(sp, sp, 16);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
FillStaticRegs();
|
||||
|
||||
@@ -231,7 +321,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
}
|
||||
|
||||
{
|
||||
SignalHandlerReturnAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
SignalHandlerReturnAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Now to get back to our old location we need to do a fault dance
|
||||
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
|
||||
@@ -239,12 +329,48 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, X86State::X86_TRAPNO_OF);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)));
|
||||
LoadConstant(w1, 0x80);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)));
|
||||
LoadConstant(x1, 0);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)));
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
ldr(x1, MemOperand(x1));
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
bind(&ThreadPauseHandler);
|
||||
ThreadPauseHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
// We will have faulted and jumped to this location at this point
|
||||
|
||||
@@ -254,7 +380,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
ldr(x2, &l_Sleep);
|
||||
blr(x2);
|
||||
|
||||
PauseReturnInstruction = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
PauseReturnInstruction = GetCursorAddress<uint64_t>();
|
||||
// Fault to start running again
|
||||
hlt(0);
|
||||
}
|
||||
@@ -275,7 +401,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
|
||||
// When the thunk itself returns, it'll do its regular return logic there
|
||||
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
CallbackPtr = Buffer->GetOffsetAddress<CPUBackend::JITCallback>(GetCursorOffset());
|
||||
CallbackPtr = GetCursorAddress<CPUBackend::JITCallback>();
|
||||
|
||||
// We expect the thunk to have previously pushed the registers it was using
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -324,28 +450,32 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
|
||||
FinalizeCode();
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
GetBuffer()->SetExecutable();
|
||||
|
||||
#if ENABLE_JITSYMBOLS
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
#endif
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(void *ucontext) {
|
||||
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
|
||||
for(int i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
}
|
||||
// TODO: Also recover FPRs, not sure where the neon context is
|
||||
// This is usually not needed
|
||||
/*
|
||||
|
||||
for(int i = 0; i < SRAFPR.size(); i++) {
|
||||
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
|
||||
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&ThreadState->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
*/
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
|
||||
@@ -3,7 +3,13 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -12,7 +18,7 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
|
||||
protected:
|
||||
void SpillSRA(void *ucontext) override;
|
||||
void SpillSRA(void *ucontext, uint32_t IgnoreMask) override;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
+412
-89
@@ -1,8 +1,24 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <bits/types/siginfo_t.h>
|
||||
#include <csignal>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -12,15 +28,19 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
|
||||
--ctx->IdleWaitRefCount;
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
|
||||
Thread->RunningEvents.ThreadSleeping = true;
|
||||
|
||||
// Go to sleep
|
||||
Thread->StartRunning.Wait();
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
++ctx->IdleWaitRefCount;
|
||||
Thread->RunningEvents.ThreadSleeping = false;
|
||||
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
void Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
// We can end up getting a signal at any point in our host state
|
||||
// Jump to a handler that saves all state so we can safely return
|
||||
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -50,13 +70,35 @@ void Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
// Set the new SP
|
||||
ArchHelpers::Context::SetSp(ucontext, NewSP);
|
||||
|
||||
SignalFrames.push(NewSP);
|
||||
// Signal frames are only used on the interpreter
|
||||
// The JITS require the stack to be setup correctly on rt_sigreturn
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
SignalFrames.push(NewSP);
|
||||
}
|
||||
|
||||
Context->Flags = 0;
|
||||
Context->FPStateLocation = 0;
|
||||
Context->UContextLocation = 0;
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
}
|
||||
|
||||
void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
LOGMAN_THROW_A(!SignalFrames.empty(), "Trying to restore a signal frame when we don't have any");
|
||||
uint64_t OldSP = SignalFrames.top();
|
||||
SignalFrames.pop();
|
||||
uint64_t OldSP{};
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
|
||||
OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(!SignalFrames.empty(), "Trying to restore a signal frame when we don't have any");
|
||||
OldSP = SignalFrames.top();
|
||||
SignalFrames.pop();
|
||||
}
|
||||
|
||||
uintptr_t NewSP = OldSP;
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
|
||||
@@ -66,83 +108,318 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
// Now restore host state
|
||||
ArchHelpers::Context::RestoreContext(ucontext, Context);
|
||||
|
||||
// Restore the previous signal state
|
||||
// This allows recursive signals to properly handle signal masking as we are walking back up the list of signals
|
||||
CTX->SignalDelegation->SetCurrentSignal(Context->Signal);
|
||||
if (Context->UContextLocation) {
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
|
||||
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
|
||||
// XXX: Unsupported since it needs state reconstruction
|
||||
// If we are in the JIT then SRA might need to be restored to values from the context
|
||||
// We can't currently support this since it might result in tearing without real state reconstruction
|
||||
}
|
||||
|
||||
if (!(Context->Flags & ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT)) {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
|
||||
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
|
||||
COPY_REG(R8);
|
||||
COPY_REG(R9);
|
||||
COPY_REG(R10);
|
||||
COPY_REG(R11);
|
||||
COPY_REG(R12);
|
||||
COPY_REG(R13);
|
||||
COPY_REG(R14);
|
||||
COPY_REG(R15);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
memcpy(Frame->State.xmm, fpstate->_xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(Context->SigInfoLocation);
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
return X86State::X86_TRAPNO_PF;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Unknown mapping, fall back to old behaviour and just pass signal
|
||||
return Signal;
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
// Always a user fault for us
|
||||
// XXX: PF_PROT and PF_WRITE
|
||||
return X86State::X86_PF_USER;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Not a page fault issue
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
StoreThreadState(Signal, ucontext);
|
||||
auto ContextBackup = StoreThreadState(Signal, ucontext);
|
||||
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
// Set the new PC
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
|
||||
uint64_t OldGuestSP = Frame->State.gregs[X86State::REG_RSP];
|
||||
uint64_t NewGuestSP = OldGuestSP;
|
||||
|
||||
if (!(GuestStack->ss_flags & SS_DISABLE)) {
|
||||
// If our guest is already inside of the alternative stack
|
||||
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
|
||||
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
|
||||
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
|
||||
if (OldGuestSP >= AltStackBase &&
|
||||
OldGuestSP <= AltStackEnd) {
|
||||
// We are already in the alt stack, the rest of the code will handle adjusting this
|
||||
}
|
||||
else {
|
||||
NewGuestSP = AltStackEnd;
|
||||
// Pulling from context here
|
||||
bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (SRAEnabled) {
|
||||
if (IsAddressInJITCode(OldPC, false)) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
// We are in a syscall, this means we are in a weird register state
|
||||
// We need to spill SRA but only some of it, since some values have already been spilled
|
||||
// Lower 16 bits tells us which registers are already spilled to the context
|
||||
// So we ignore spilling those ones
|
||||
uint16_t NumRegisters = std::popcount(Frame->InSyscallInfo & 0xFFFF);
|
||||
if (NumRegisters >= 4) {
|
||||
// Unhandled case
|
||||
IgnoreMask = 0;
|
||||
}
|
||||
else {
|
||||
IgnoreMask = Frame->InSyscallInfo & 0xFFFF;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We must spill everything
|
||||
IgnoreMask = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(ucontext, IgnoreMask);
|
||||
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
|
||||
} else {
|
||||
if (!IsAddressInJITCode(OldPC, true)) {
|
||||
// This is likely to cause issues but in some cases it isn't fatal
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
// Only throw a log message in this case
|
||||
if constexpr (false) {
|
||||
// XXX: Messages in the signal handler can cause us to crash
|
||||
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Back up past the redzone, which is 128bytes
|
||||
// Don't need this offset if we aren't going to be putting siginfo in to it
|
||||
NewGuestSP -= 128;
|
||||
// altstack is only used if the signal handler was setup with SA_ONSTACK
|
||||
if (GuestAction->sa_flags & SA_ONSTACK) {
|
||||
// Additionally the altstack is only used if the enabled (SS_DISABLE flag is not set)
|
||||
if (!(GuestStack->ss_flags & SS_DISABLE)) {
|
||||
// If our guest is already inside of the alternative stack
|
||||
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
|
||||
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
|
||||
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
|
||||
if (OldGuestSP >= AltStackBase &&
|
||||
OldGuestSP <= AltStackEnd) {
|
||||
// We are already in the alt stack, the rest of the code will handle adjusting this
|
||||
}
|
||||
else {
|
||||
NewGuestSP = AltStackEnd;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Is64BitMode) {
|
||||
// Back up past the redzone, which is 128bytes
|
||||
// 32-bit doesn't have a redzone
|
||||
NewGuestSP -= 128;
|
||||
}
|
||||
|
||||
// siginfo_t
|
||||
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
|
||||
|
||||
if (GuestAction->sa_flags & SA_SIGINFO &&
|
||||
!(HostSigInfo->si_code == SI_QUEUE || // If the siginfo comes from sigqueue or user then we don't need to check
|
||||
HostSigInfo->si_code == SI_USER)) {
|
||||
if (SRAEnabled) {
|
||||
if (!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
LOGMAN_THROW_A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
|
||||
} else {
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(ucontext);
|
||||
}
|
||||
}
|
||||
// Backup where we think the RIP currently is
|
||||
ContextBackup->OriginalRIP = Frame->State.rip;
|
||||
|
||||
if (GuestAction->sa_flags & SA_SIGINFO) {
|
||||
// Setup ucontext a bit
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
if (Is64BitMode) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::_libc_fpstate));
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(siginfo_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86_64::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(UContextLocation);
|
||||
siginfo_t *guest_siginfo = reinterpret_cast<siginfo_t*>(SigInfoLocation);
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags |= FEXCore::x86_64::UC_FP_XSTATE;
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = &guest_uctx->__fpregs_mem;
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Signal;
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
|
||||
|
||||
@@ -167,15 +444,15 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
#undef COPY_REG
|
||||
|
||||
// Copy float registers
|
||||
memcpy(guest_uctx->__fpregs_mem._st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
memcpy(guest_uctx->__fpregs_mem._xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
guest_uctx->__fpregs_mem.fcw = Frame->State.FCW;
|
||||
guest_uctx->__fpregs_mem.ftw = Frame->State.FTW;
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.FTW;
|
||||
|
||||
// Reconstruct FSW
|
||||
guest_uctx->__fpregs_mem.fsw =
|
||||
fpstate->fsw =
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
@@ -187,36 +464,52 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
|
||||
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
|
||||
}
|
||||
else {
|
||||
// XXX: 32bit Support
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::siginfo_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(UContextLocation);
|
||||
FEXCore::x86::siginfo_t *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(SigInfoLocation);
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags |= FEXCore::x86::UC_FP_XSTATE;
|
||||
guest_uctx->uc_flags = FEXCore::x86::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(reinterpret_cast<uint64_t>(&guest_uctx->__fpregs_mem));
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(FPStateLocation);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Signal;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = 0;
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
@@ -236,20 +529,20 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
#undef COPY_REG
|
||||
|
||||
// Copy float registers
|
||||
memcpy(guest_uctx->__fpregs_mem._st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
if (0) {
|
||||
// XXX: Handle XMM
|
||||
// memcpy(guest_uctx->__fpregs_mem._xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
guest_uctx->__fpregs_mem.status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
}
|
||||
else {
|
||||
guest_uctx->__fpregs_mem.status = FEXCore::x86::fpstate_magic::MAGIC_FPU;
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
guest_uctx->__fpregs_mem.fcw = Frame->State.FCW;
|
||||
guest_uctx->__fpregs_mem.ftw = Frame->State.FTW;
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.FTW;
|
||||
// Reconstruct FSW
|
||||
guest_uctx->__fpregs_mem.fsw =
|
||||
fpstate->fsw =
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
@@ -264,11 +557,16 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
// These three elements are in every siginfo
|
||||
guest_siginfo->si_signo = HostSigInfo->si_signo;
|
||||
guest_siginfo->si_errno = HostSigInfo->si_errno;
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
case SIGBUS:
|
||||
// Macro expansion to get the si_addr
|
||||
// This is the address trying to be accessed, not the RIP
|
||||
guest_siginfo->_sifields._sigfault.addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_addr));
|
||||
break;
|
||||
case SIGFPE:
|
||||
case SIGILL:
|
||||
// Macro expansion to get the si_addr
|
||||
// Can't really give a real result here. Pull from the context for now
|
||||
guest_siginfo->_sifields._sigfault.addr = Frame->State.rip;
|
||||
@@ -280,10 +578,15 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_siginfo->_sifields._sigchld.utime = HostSigInfo->si_utime;
|
||||
guest_siginfo->_sifields._sigchld.stime = HostSigInfo->si_stime;
|
||||
break;
|
||||
default:
|
||||
// Hope for the best, most things just copy over
|
||||
memcpy(&guest_siginfo->_sifields, &HostSigInfo->_sifields, sizeof(siginfo_t));
|
||||
break;
|
||||
case SIGALRM:
|
||||
case SIGVTALRM:
|
||||
guest_siginfo->_sifields._timer.tid = HostSigInfo->si_timerid;
|
||||
guest_siginfo->_sifields._timer.overrun = HostSigInfo->si_overrun;
|
||||
guest_siginfo->_sifields._timer.sigval.sival_int = HostSigInfo->si_int;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled siginfo_t for signal: {}\n", Signal);
|
||||
break;
|
||||
}
|
||||
|
||||
NewGuestSP -= 4;
|
||||
@@ -297,7 +600,7 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.sigaction);
|
||||
}
|
||||
else {
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
if (!Is64BitMode) {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = Signal;
|
||||
}
|
||||
@@ -305,21 +608,29 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.handler);
|
||||
}
|
||||
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Frame->State.gregs[X86State::REG_RDI] = Signal;
|
||||
if (Is64BitMode) {
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDI] = Signal;
|
||||
|
||||
// Set up the new SP for stack handling
|
||||
NewGuestSP -= 8;
|
||||
*(uint64_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
|
||||
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
|
||||
*(uint64_t*)NewGuestSP = SignalReturn;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
else {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
|
||||
LOGMAN_THROW_A(CTX->X86CodeGen.SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
|
||||
*(uint32_t*)NewGuestSP = SignalReturn;
|
||||
LOGMAN_THROW_A_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
// This state will be restored on rt_sigreturn
|
||||
memset(Frame->State.xmm, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
|
||||
Frame->State.FCW = 0x37F;
|
||||
Frame->State.FTW = 0xFFFF;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -360,14 +671,12 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
}
|
||||
|
||||
// Set the new PC
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
@@ -395,11 +704,21 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
}
|
||||
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (ThreadState->RunningEvents.ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
++ThreadState->CTX->IdleWaitRefCount;
|
||||
}
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
@@ -440,15 +759,19 @@ void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) const {
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher, bool IncludeCompileService) const {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (IncludeDispatcher) {
|
||||
return IsAddressInDispatcher(Address);
|
||||
if (IncludeDispatcher && IsAddressInDispatcher(Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (IncludeCompileService && ThreadState->CompileService && ThreadState->CompileService->IsAddressInJITCode(Address)) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1,11 +1,25 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <bits/types/stack_t.h>
|
||||
#include <cstdint>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestSigAction;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -34,11 +48,20 @@ public:
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
struct SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
} SynchronousFaultData;
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
@@ -54,7 +77,7 @@ public:
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const;
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const;
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
@@ -64,12 +87,12 @@ protected:
|
||||
: CTX {ctx}
|
||||
, ThreadState {Thread} {}
|
||||
|
||||
void StoreThreadState(int Signal, void *ucontext);
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(int Signal, void *ucontext);
|
||||
void RestoreThreadState(void *ucontext);
|
||||
std::stack<uint64_t> SignalFrames;
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
|
||||
bool SRAEnabled = false;
|
||||
virtual void SpillSRA(void *ucontext) {}
|
||||
virtual void SpillSRA(void *ucontext, uint32_t IgnoreMask) {}
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
|
||||
@@ -3,10 +3,22 @@
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include "xbyak/xbyak.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
@@ -130,7 +142,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
je(NoBlock);
|
||||
|
||||
// Update L1
|
||||
|
||||
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
@@ -263,10 +275,33 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
{
|
||||
// Signal return handler
|
||||
SignalHandlerReturnAddress = getCurr<uint64_t>();
|
||||
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = getCurr<uint64_t>();
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
add(byte [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)], 1);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)], X86State::X86_TRAPNO_OF);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)], 0);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)], 0x80);
|
||||
|
||||
hlt();
|
||||
}
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
@@ -295,10 +330,13 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
Start = reinterpret_cast<uint64_t>(getCode());
|
||||
End = Start + getSize();
|
||||
|
||||
#if ENABLE_JITSYMBOLS
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
|
||||
#endif
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
|
||||
}
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
|
||||
@@ -2,11 +2,17 @@
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
|
||||
|
||||
+152
-57
@@ -7,14 +7,18 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
@@ -123,6 +127,54 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
return (*GPRs)[(REX << 3) | bits];
|
||||
}
|
||||
|
||||
static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
using GPRArray = std::array<uint32_t, 16>;
|
||||
|
||||
static constexpr GPRArray GPRIndexes = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_RBX,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
FEXCore::X86State::REG_RBP,
|
||||
FEXCore::X86State::REG_RSI,
|
||||
FEXCore::X86State::REG_RDI,
|
||||
FEXCore::X86State::REG_R8,
|
||||
FEXCore::X86State::REG_R9,
|
||||
FEXCore::X86State::REG_R10,
|
||||
FEXCore::X86State::REG_R11,
|
||||
FEXCore::X86State::REG_R12,
|
||||
FEXCore::X86State::REG_R13,
|
||||
FEXCore::X86State::REG_R14,
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray XMMIndexes = {
|
||||
FEXCore::X86State::REG_XMM_0,
|
||||
FEXCore::X86State::REG_XMM_1,
|
||||
FEXCore::X86State::REG_XMM_2,
|
||||
FEXCore::X86State::REG_XMM_3,
|
||||
FEXCore::X86State::REG_XMM_4,
|
||||
FEXCore::X86State::REG_XMM_5,
|
||||
FEXCore::X86State::REG_XMM_6,
|
||||
FEXCore::X86State::REG_XMM_7,
|
||||
FEXCore::X86State::REG_XMM_8,
|
||||
FEXCore::X86State::REG_XMM_9,
|
||||
FEXCore::X86State::REG_XMM_10,
|
||||
FEXCore::X86State::REG_XMM_11,
|
||||
FEXCore::X86State::REG_XMM_12,
|
||||
FEXCore::X86State::REG_XMM_13,
|
||||
FEXCore::X86State::REG_XMM_14,
|
||||
FEXCore::X86State::REG_XMM_15,
|
||||
};
|
||||
|
||||
if (HasXMM) {
|
||||
return XMMIndexes[vvvv];
|
||||
} else {
|
||||
return GPRIndexes[vvvv];
|
||||
}
|
||||
}
|
||||
|
||||
Decoder::Decoder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN } {
|
||||
@@ -140,7 +192,7 @@ Decoder::~Decoder() {
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_A(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
@@ -157,7 +209,7 @@ uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
}
|
||||
|
||||
if (Size > sizeof(uint64_t)) {
|
||||
LOGMAN_MSG_A("Unknown data size to read");
|
||||
LOGMAN_MSG_A_FMT("Unknown data size to read");
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -299,10 +351,9 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
|
||||
uint64_t Literal {0};
|
||||
LOGMAN_THROW_A(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
Literal = ReadData(Displacement);
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
@@ -312,8 +363,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
if (ModRM.rm == 0b101) {
|
||||
// 32bit Displacement
|
||||
uint32_t Literal;
|
||||
Literal = ReadData(4);
|
||||
const uint32_t Literal = ReadData(4);
|
||||
|
||||
Operand->Type = DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value.u = Literal;
|
||||
@@ -326,8 +376,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
else {
|
||||
uint8_t DisplacementSize = ModRM.mod == 1 ? 1 : 4;
|
||||
uint32_t Literal{};
|
||||
Literal = ReadData(DisplacementSize);
|
||||
uint32_t Literal = ReadData(DisplacementSize);
|
||||
if (DisplacementSize == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
@@ -339,32 +388,33 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op) {
|
||||
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options) {
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::D("Legacy Prefix");
|
||||
LogMan::Msg::DFmt("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
|
||||
uint8_t DestSize{};
|
||||
bool HasWideningDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST;
|
||||
bool HasNarrowingDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST;
|
||||
const bool HasWideningDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST) != 0 ||
|
||||
(Options.w && CTX->Config.Is64BitMode);
|
||||
const bool HasNarrowingDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
|
||||
|
||||
bool HasXMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
@@ -397,8 +447,8 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
// New instruction size decoding
|
||||
{
|
||||
// Decode destinations first
|
||||
uint32_t DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
|
||||
uint32_t SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
|
||||
const auto DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
|
||||
const auto SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
|
||||
|
||||
if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_8BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_8BIT);
|
||||
@@ -482,7 +532,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_A(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
@@ -542,6 +592,13 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
|
||||
size_t CurrentSrc = 0;
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
|
||||
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest))
|
||||
@@ -554,6 +611,13 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
@@ -567,8 +631,14 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
@@ -593,7 +663,8 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A(Bytes == 0, "Inst at 0x%lx: 0x%04x '%s' Had an instruction of size %d with %d remaining", DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
}
|
||||
@@ -604,21 +675,22 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::D("Legacy Prefix");
|
||||
LogMan::Msg::DFmt("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
"REX PREFIX should have been decoded before this!");
|
||||
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
@@ -674,7 +746,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_A(Field != 255, "Invalid field selected!");
|
||||
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -696,20 +768,36 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
}
|
||||
else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
|
||||
uint16_t map_select = 1;
|
||||
uint16_t pp = 0;
|
||||
const uint8_t Byte1 = ReadByte();
|
||||
DecodedHeader options{};
|
||||
|
||||
uint8_t Byte1 = ReadByte();
|
||||
if ((Byte1 & 0b10000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.R shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
}
|
||||
|
||||
if (Op == 0xC5) { // Two byte VEX
|
||||
pp = Byte1 & 0b11;
|
||||
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
|
||||
}
|
||||
else { // 0xC4 = Three byte VEX
|
||||
uint8_t Byte2 = ReadByte();
|
||||
const uint8_t Byte2 = ReadByte();
|
||||
pp = Byte2 & 0b11;
|
||||
map_select = Byte1 & 0b11111;
|
||||
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
}
|
||||
if (CTX->Config.Is64BitMode && (Byte1 & 0b00100000) == 0) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
LogMan::Msg::E("We don't understand a map_select of: %d", map_select);
|
||||
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -723,6 +811,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
|
||||
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 &&
|
||||
LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
|
||||
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
@@ -734,12 +823,14 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
|
||||
#undef OPD
|
||||
return NormalOp(&VEXTableGroupOps[Op], Op);
|
||||
return NormalOp(&VEXTableGroupOps[Op], Op, options);
|
||||
} else {
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
}
|
||||
else
|
||||
return NormalOp(LocalInfo, Op);
|
||||
}
|
||||
else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
|
||||
/* uint8_t P1 = */ ReadByte();
|
||||
/* uint8_t P2 = */ ReadByte();
|
||||
/* uint8_t P3 = */ ReadByte();
|
||||
@@ -789,14 +880,20 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = 1;
|
||||
constexpr uint16_t PF_38_F2 = 2;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
uint16_t Prefix = PF_38_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0xF2) // REPNE
|
||||
Prefix = PF_38_F2;
|
||||
else if (DecodeInst->LastEscapePrefix == 0x66) // Operand Size
|
||||
Prefix = PF_38_66;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
|
||||
Prefix |= PF_38_66;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
|
||||
Prefix |= PF_38_F2;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
|
||||
Prefix |= PF_38_F3;
|
||||
}
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
|
||||
@@ -911,7 +1008,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
auto Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
LOGMAN_THROW_A(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
|
||||
// Widening displacement
|
||||
@@ -964,13 +1061,13 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
|
||||
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
// Target offset is PC + InstSize + Literal
|
||||
LOGMAN_THROW_A(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
break;
|
||||
}
|
||||
case 0xE9:
|
||||
case 0xEB: // Both are unconditional JMP instructions
|
||||
LOGMAN_THROW_A(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
Conditional = false;
|
||||
break;
|
||||
@@ -1036,7 +1133,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1050,7 +1147,6 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
EntryPoint = PC;
|
||||
InstStream = _InstStream;
|
||||
|
||||
bool ErrorDuringDecoding = false;
|
||||
uint64_t TotalInstructions{};
|
||||
|
||||
// If we don't have symbols available then we become a bit optimistic about multiblock ranges
|
||||
@@ -1082,20 +1178,15 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::D("Couldn't Decode something at 0x%lx, Started at 0x%lx", PC + PCOffset, PC);
|
||||
if (Blocks.size() == 1) {
|
||||
return false;
|
||||
}
|
||||
LOGMAN_THROW_A(Blocks.size() != 1, "Decode Error in entry block");
|
||||
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", PC + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
if (ErrorDuringDecoding && Blocks.size() != 1) {
|
||||
ErrorDuringDecoding = false;
|
||||
}
|
||||
break;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
DecodeInst->InstSize = 0;
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
@@ -1104,6 +1195,11 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
++BlockNumberOfInstructions;
|
||||
++DecodedSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool CanContinue = false;
|
||||
if (!(DecodeInst->TableInfo->Flags &
|
||||
(FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
|
||||
@@ -1148,7 +1244,6 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
std::sort(Blocks.begin(), Blocks.end(), [](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
|
||||
return a.Entry < b.Entry;
|
||||
});
|
||||
return !ErrorDuringDecoding;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+17
-6
@@ -2,12 +2,12 @@
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <set>
|
||||
#include <stack>
|
||||
#include <stddef.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -27,7 +27,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
~Decoder();
|
||||
bool DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
@@ -35,10 +35,17 @@ public:
|
||||
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
private:
|
||||
// To pass any information from instruction prefixes
|
||||
// down into the actual instruction handling machinery.
|
||||
struct DecodedHeader {
|
||||
uint8_t vvvv; // Encoded operand in a VEX prefix.
|
||||
bool w; // VEX.W bit.
|
||||
};
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI{};
|
||||
|
||||
@@ -50,7 +57,8 @@ private:
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
uint64_t ReadData(uint8_t Size);
|
||||
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
|
||||
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
|
||||
|
||||
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
|
||||
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
@@ -83,11 +91,14 @@ private:
|
||||
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
|
||||
const std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
|
||||
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_64,
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_16,
|
||||
};
|
||||
|
||||
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
}
|
||||
+258
-89
@@ -8,30 +8,42 @@ $end_info$
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include "Common/NetStream.h"
|
||||
#include <vector>
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <fmt/format.h>
|
||||
#include <fstream>
|
||||
#include <fmt/format.h>
|
||||
#include <netdb.h>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <string_view>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/types.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "GdbServer.h"
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
|
||||
namespace FEXCore
|
||||
{
|
||||
@@ -462,6 +474,48 @@ std::string buildTargetXML() {
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
std::string buildMemoryMap() {
|
||||
std::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
|
||||
xml << "<!DOCTYPE memory-map>\n";
|
||||
xml << "<memory-map>\n";
|
||||
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
|
||||
while (std::getline(fs, Line)) {
|
||||
if (fs.eof()) break;
|
||||
uint64_t Begin, End;
|
||||
char r,w,x,p;
|
||||
if (sscanf(Line.c_str(), "%lx-%lx %c%c%c%c", &Begin, &End, &r, &w, &x, &p) == 6) {
|
||||
xml << "<memory type=\"ram\" start=\"0x" << std::hex << Begin << "\" length=\"0x" << (End - Begin) << "\"/>\n";
|
||||
}
|
||||
}
|
||||
|
||||
xml << "</memory-map>\n";
|
||||
|
||||
xml << std::flush;
|
||||
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
std::string buildOSData() {
|
||||
std::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
|
||||
xml << "<!DOCTYPE target SYSTEM \"osdata.dtd\">\n";
|
||||
xml << "<osdata type=\"processes\">";
|
||||
// XXX
|
||||
xml << "</osdata>";
|
||||
|
||||
xml << std::flush;
|
||||
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
std::string object;
|
||||
std::string rw;
|
||||
@@ -528,11 +582,11 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
|
||||
ThreadString.clear();
|
||||
std::ostringstream ss;
|
||||
ss << "<?xml version=\"1.0\?>\n";
|
||||
ss << "<?xml version=\"1.0\"?>\n";
|
||||
ss << "<threads>\n";
|
||||
for (size_t i = 0; i < Threads->size(); ++i) {
|
||||
auto Thread = Threads->at(i);
|
||||
ss << "\t<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\" core=\"" << i << "\" name=\"" << getThreadName(Thread->ThreadManager.GetTID()) << "\">\n";
|
||||
for (auto &Thread : *Threads) {
|
||||
// Thread id is in hex without 0x prefix
|
||||
ss << "\t<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\" name=\"" << getThreadName(Thread->ThreadManager.GetTID()) << "\">\n";
|
||||
ss << "\t</thread>\n";
|
||||
}
|
||||
|
||||
@@ -541,8 +595,22 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
ThreadString = ss.str();
|
||||
}
|
||||
|
||||
return {encode(ThreadString.substr(offset, length)), HandledPacketType::TYPE_ACK};
|
||||
return {encode(ThreadString), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
if (object == "memory-map") {
|
||||
if (offset == 0) {
|
||||
MemoryMapString = buildMemoryMap();
|
||||
}
|
||||
return {encode(MemoryMapString), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (object == "osdata") {
|
||||
if (offset == 0) {
|
||||
OSDataString = buildOSData();
|
||||
}
|
||||
return {encode(OSDataString), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
@@ -632,12 +700,82 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet)
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const std::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
|
||||
if (match("qSupported")) {
|
||||
return {"PacketSize=5000;xmlRegisters=i386;qXfer:exec-file:read+;qXfer:features:read+;", HandledPacketType::TYPE_ACK};
|
||||
const auto split = [](const std::string &Str, char deliminator) -> std::vector<std::string> {
|
||||
std::vector<std::string> Elements;
|
||||
std::istringstream Input(Str);
|
||||
for (std::string line;
|
||||
std::getline(Input, line);
|
||||
Elements.emplace_back(line));
|
||||
return Elements;
|
||||
};
|
||||
|
||||
if (match("QNonStop:")) {
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("QNonStop:").size());
|
||||
ss.get(); // discard colon
|
||||
ss >> NonStopMode;
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qSupported:")) {
|
||||
// eg: qSupported:multiprocess+;swbreak+;hwbreak+;qRelocInsn+;fork-events+;vfork-events+;exec-events+;vContSupported+;QThreadEvents+;no-resumed+;memory-tagging+;xmlRegisters=i386
|
||||
auto Features = split(packet.substr(strlen("qSupported:")), ';');
|
||||
|
||||
// For feature documentation
|
||||
// https://sourceware.org/gdb/current/onlinedocs/gdb/General-Query-Packets.html#qSupported
|
||||
std::string SupportedFeatures{};
|
||||
|
||||
// Required features
|
||||
SupportedFeatures += "PacketSize=5000;";
|
||||
SupportedFeatures += "xmlRegisters=i386;";
|
||||
|
||||
// XXX: Not yet supported, would be easy
|
||||
// SupportedFeatures += "qXfer:auxv-file:read+";
|
||||
SupportedFeatures += "qXfer:exec-file:read+;";
|
||||
SupportedFeatures += "qXfer:features:read+;";
|
||||
// XXX: Requires parsing the ELF and watching the library list
|
||||
// SupportedFeatures += "qXfer:libraries:read+;";
|
||||
SupportedFeatures += "qXfer:memory-map:read+;";
|
||||
SupportedFeatures += "qXfer:siginfo:read+;";
|
||||
SupportedFeatures += "qXfer:siginfo:write+;";
|
||||
// XXX: Allowing this causes GDB to crash
|
||||
SupportedFeatures += "qXfer:threads:read+;";
|
||||
// QCatchSignals
|
||||
// QPassSignals
|
||||
SupportedFeatures += "QNonStop+;";
|
||||
|
||||
SupportedFeatures += "qXfer:osdata:read+;";
|
||||
|
||||
// Causes GDB to crash?
|
||||
// SupportedFeatures += "QStartNoAckMode+;";
|
||||
|
||||
for (auto &Feature : Features) {
|
||||
|
||||
if (MatchStr(Feature, "swbreak+")) {
|
||||
SupportedFeatures += "swbreak+;";
|
||||
}
|
||||
if (MatchStr(Feature, "hwbreak+")) {
|
||||
SupportedFeatures += "hwbreak+;";
|
||||
}
|
||||
if (MatchStr(Feature, "vContSupported+")) {
|
||||
SupportedFeatures += "vContSupported+;";
|
||||
}
|
||||
|
||||
// Unsupported:
|
||||
// multiprocess
|
||||
// qRelocInsn
|
||||
// fork-events
|
||||
// vfork-events
|
||||
// exec-events
|
||||
// QThreadEvents
|
||||
// no-resumed
|
||||
// memory-tagging
|
||||
}
|
||||
return {SupportedFeatures, HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qAttached")) {
|
||||
return {"1", HandledPacketType::TYPE_ACK}; // We don't currently support launching executables from gdb.
|
||||
return {"tnotrun:0", HandledPacketType::TYPE_ACK}; // We don't currently support launching executables from gdb.
|
||||
}
|
||||
if (match("qXfer")) {
|
||||
return handleXfer(packet);
|
||||
@@ -656,7 +794,10 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
ss << "m";
|
||||
for (size_t i = 0; i < Threads->size(); ++i) {
|
||||
auto Thread = Threads->at(i);
|
||||
ss << std::hex << Thread->ThreadManager.TID << ",";
|
||||
ss << std::hex << Thread->ThreadManager.TID;
|
||||
if (i != (Threads->size() - 1)) {
|
||||
ss << ",";
|
||||
}
|
||||
}
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -689,6 +830,30 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid) {
|
||||
switch (action) {
|
||||
case 'c': {
|
||||
CTX->Run();
|
||||
CTX->WaitForThreadsToRun();
|
||||
return {"", HandledPacketType::TYPE_ONLYACK};
|
||||
}
|
||||
case 's': {
|
||||
CTX->Step();
|
||||
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
|
||||
auto str = fmt::format("T05thread:{:02x};core:2c;", getpid());
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 't':
|
||||
// This thread isn't part of the thread pool
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
default:
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
const auto match = [&](const std::string& str) -> std::optional<std::istringstream> {
|
||||
if (packet.rfind(str, 0) == 0) {
|
||||
@@ -751,44 +916,25 @@ GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
return {F_data(ret, data), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if ((ss = match("vCont?"))) {
|
||||
return {"vCont;c;C;t;s;S;r", HandledPacketType::TYPE_ACK}; // We support continue, step and terminate
|
||||
// FIXME: We also claim to support continue with signal... because it's compulsory
|
||||
return {"vCont;c;t;s;r", HandledPacketType::TYPE_ACK}; // We support continue, step and terminate
|
||||
// FIXME: We also claim to support continue with signal... because it's compulsory
|
||||
}
|
||||
if ((ss = match("vCont;"))) {
|
||||
char action;
|
||||
int thread;
|
||||
char action{};
|
||||
int thread{};
|
||||
|
||||
action = ss->get();
|
||||
action = ss->get();
|
||||
|
||||
if (ss->peek() == ':') {
|
||||
ss->get();
|
||||
*ss >> std::hex >> thread;
|
||||
}
|
||||
if (ss->peek() == ':') {
|
||||
ss->get();
|
||||
*ss >> std::hex >> thread;
|
||||
}
|
||||
|
||||
if (ss->fail()) {
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
switch (action) {
|
||||
case 'c': {
|
||||
CTX->Run();
|
||||
return {"", HandledPacketType::TYPE_ONLYACK};
|
||||
}
|
||||
case 's': {
|
||||
CTX->Step();
|
||||
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
|
||||
auto str = fmt::format("T05thread:{:02x};core:2c;", getpid());
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 't':
|
||||
// This thread isn't part of the thread pool
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
default:
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (ss->fail()) {
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
return ThreadAction(action, thread);
|
||||
}
|
||||
return {"", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -824,7 +970,8 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &packet) {
|
||||
auto ss = std::istringstream(packet);
|
||||
|
||||
bool Set{};
|
||||
// Don't do anything with set breakpoints yet
|
||||
[[maybe_unused]] bool Set{};
|
||||
uint64_t Addr;
|
||||
uint64_t Type;
|
||||
Set = ss.get() == 'Z';
|
||||
@@ -843,11 +990,24 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet)
|
||||
// Indicates the reason that the thread has stopped
|
||||
// Behaviour changes if the target is in non-stop mode
|
||||
// Binja doesn't support S response here
|
||||
//return {"S00", HandledPacketType::TYPE_ACK};
|
||||
auto str = fmt::format("T00thread:{:02x};core:2c;", getpid());
|
||||
auto str = fmt::format("T00thread:{:x};", getpid());
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 'c':
|
||||
// Continue
|
||||
CTX->Run();
|
||||
CTX->WaitForThreadsToRun();
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
case 'D':
|
||||
// Detach
|
||||
// Ensure the threads are back in running state on detach
|
||||
CTX->Run();
|
||||
CTX->WaitForThreadsToRun();
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
case 'g':
|
||||
// We might be running while we try reading
|
||||
// Pause up front
|
||||
CTX->Pause();
|
||||
return {readRegs(), HandledPacketType::TYPE_ACK};
|
||||
case 'p':
|
||||
return readReg(packet);
|
||||
@@ -864,6 +1024,8 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet)
|
||||
case '!': // Enable extended mode
|
||||
case 'T': // Is a thread alive?
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
case 's': // Step
|
||||
return ThreadAction('s', 0);
|
||||
case 'Z': // Inserts breakpoint or watchpoint
|
||||
return handleBreakpoint(packet);
|
||||
case 'k': // Kill the process
|
||||
@@ -897,6 +1059,8 @@ void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
}
|
||||
|
||||
void GdbServer::GdbServerLoop() {
|
||||
OpenListenSocket();
|
||||
|
||||
while (!CTX->CoreShuttingDown.load()) {
|
||||
CommsStream = OpenSocket();
|
||||
|
||||
@@ -942,6 +1106,8 @@ void GdbServer::GdbServerLoop() {
|
||||
CommsStream.reset();
|
||||
}
|
||||
}
|
||||
|
||||
close(ListenSocket);
|
||||
}
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::GdbServer *This = reinterpret_cast<FEXCore::GdbServer*>(Arg);
|
||||
@@ -950,49 +1116,52 @@ static void* ThreadHandler(void *Arg) {
|
||||
}
|
||||
|
||||
void GdbServer::StartThread() {
|
||||
gdbServerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
gdbServerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void GdbServer::OpenListenSocket() {
|
||||
// open socket
|
||||
struct addrinfo hints, *res;
|
||||
|
||||
memset(&hints, 0, sizeof(hints));
|
||||
hints.ai_family = AF_UNSPEC;
|
||||
hints.ai_socktype = SOCK_STREAM;
|
||||
hints.ai_flags = AI_PASSIVE;
|
||||
|
||||
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
|
||||
perror("getaddrinfo");
|
||||
}
|
||||
|
||||
int on = 1;
|
||||
|
||||
ListenSocket = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
|
||||
if (ListenSocket < 0) {
|
||||
perror("socket");
|
||||
}
|
||||
if(setsockopt(ListenSocket, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
|
||||
perror("setsockopt");
|
||||
close(ListenSocket);
|
||||
}
|
||||
|
||||
if (bind(ListenSocket, res->ai_addr, res->ai_addrlen) < 0) {
|
||||
perror("bind");
|
||||
close(ListenSocket);
|
||||
}
|
||||
|
||||
listen(ListenSocket, 1);
|
||||
}
|
||||
|
||||
std::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
// open socket
|
||||
int sockfd, new_fd;
|
||||
// Block until a connection arrives
|
||||
struct sockaddr_storage their_addr;
|
||||
socklen_t addr_size;
|
||||
|
||||
struct addrinfo hints, *res;
|
||||
struct sockaddr_storage their_addr;
|
||||
socklen_t addr_size;
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
|
||||
memset(&hints, 0, sizeof(hints));
|
||||
hints.ai_family = AF_UNSPEC;
|
||||
hints.ai_socktype = SOCK_STREAM;
|
||||
hints.ai_flags = AI_PASSIVE;
|
||||
|
||||
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
|
||||
perror("getaddrinfo");
|
||||
}
|
||||
|
||||
int on = 1;
|
||||
|
||||
sockfd = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
|
||||
if (sockfd < 0) {
|
||||
perror("socket");
|
||||
}
|
||||
if(setsockopt(sockfd, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
|
||||
perror("setsockopt");
|
||||
}
|
||||
|
||||
if (bind(sockfd, res->ai_addr, res->ai_addrlen) < 0) {
|
||||
perror("bind");
|
||||
}
|
||||
|
||||
// Block until a connection arrives
|
||||
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
listen(sockfd, 1);
|
||||
|
||||
new_fd = accept(sockfd, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
return std::make_unique<NetStream>(new_fd);
|
||||
return std::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
}
|
||||
|
||||
|
||||
|
||||
+17
-6
@@ -5,18 +5,21 @@ $end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <mutex>
|
||||
#include <thread>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Common/NetStream.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
class GdbServer {
|
||||
public:
|
||||
GdbServer(FEXCore::Context::Context *ctx);
|
||||
@@ -27,6 +30,7 @@ public:
|
||||
private:
|
||||
void Break(int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
std::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
std::string ReadPacket(std::iostream &stream);
|
||||
@@ -57,6 +61,8 @@ private:
|
||||
HandledPacketType handleBreakpoint(const std::string &packet);
|
||||
HandledPacketType handleProgramOffsets();
|
||||
|
||||
HandledPacketType ThreadAction(char action, uint32_t tid);
|
||||
|
||||
std::string readRegs();
|
||||
HandledPacketType readReg(const std::string& packet);
|
||||
|
||||
@@ -66,8 +72,13 @@ private:
|
||||
std::mutex sendMutex;
|
||||
bool SettingNoAckMode{false};
|
||||
bool NoAckMode{false};
|
||||
bool NonStopMode{false};
|
||||
std::string ThreadString{};
|
||||
std::string MemoryMapString{};
|
||||
std::string OSDataString{};
|
||||
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
};
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
@@ -13,14 +14,103 @@
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
// Log2 of the blocksize in 32-bit words
|
||||
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static uint32_t GetDCZID() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], DCZID_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static uint32_t GetFPCR() {
|
||||
uint64_t Result{};
|
||||
__asm ("mrs %[Res], FPCR"
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static void SetFPCR(uint64_t Value) {
|
||||
__asm ("msr FPCR, %[Value]"
|
||||
:: [Value] "r" (Value));
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetDCZID() {
|
||||
// Return unsupported
|
||||
return DCZID_DZP_MASK;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile ("mrs %[ctr], ctr_el0"
|
||||
: [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
#endif
|
||||
#ifdef _M_X86_64
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
(1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
SetFPCR(FPCR);
|
||||
FPCR = GetFPCR();
|
||||
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
uint32_t DCZID_Log2 = DCZID & DCZID_BS_MASK;
|
||||
uint32_t DCZID_Bytes = (1 << DCZID_Log2) * sizeof(uint32_t);
|
||||
// If the DC ZVA size matches the emulated cache line size
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,9 +1,27 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore {
|
||||
class HostFeatures final {
|
||||
public:
|
||||
HostFeatures();
|
||||
|
||||
/**
|
||||
* @brief Backend features that change how codegen is generated from IR
|
||||
*
|
||||
* Specifically things that affect the IR->Codegen process
|
||||
* Not the x86->IR process
|
||||
*/
|
||||
uint32_t DCacheLineSize{};
|
||||
uint32_t ICacheLineSize{};
|
||||
bool SupportsAES{};
|
||||
bool SupportsCRC{};
|
||||
bool SupportsCLZERO{};
|
||||
bool SupportsAtomics{};
|
||||
bool SupportsRCPC{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
bool SupportsFloatExceptions{};
|
||||
};
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,778 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#ifdef _M_X86_64
|
||||
uint8_t AtomicFetchNeg(uint8_t *Addr) {
|
||||
using Type = uint8_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
uint16_t AtomicFetchNeg(uint16_t *Addr) {
|
||||
using Type = uint16_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
uint32_t AtomicFetchNeg(uint32_t *Addr) {
|
||||
using Type = uint32_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
uint64_t AtomicFetchNeg(uint64_t *Addr) {
|
||||
using Type = uint64_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T AtomicCompareAndSwap(T expected, T desired, T *addr)
|
||||
{
|
||||
std::atomic<T> *MemData = reinterpret_cast<std::atomic<T>*>(addr);
|
||||
|
||||
T Src1 = expected;
|
||||
T Src2 = desired;
|
||||
|
||||
T Expected = Src1;
|
||||
bool Result = MemData->compare_exchange_strong(Expected, Src2);
|
||||
|
||||
return Result ? Src1 : Expected;
|
||||
}
|
||||
|
||||
template uint8_t AtomicCompareAndSwap<uint8_t>(uint8_t expected, uint8_t desired, uint8_t *addr);
|
||||
template uint16_t AtomicCompareAndSwap<uint16_t>(uint16_t expected, uint16_t desired, uint16_t *addr);
|
||||
template uint32_t AtomicCompareAndSwap<uint32_t>(uint32_t expected, uint32_t desired, uint32_t *addr);
|
||||
template uint64_t AtomicCompareAndSwap<uint64_t>(uint64_t expected, uint64_t desired, uint64_t *addr);
|
||||
|
||||
#else
|
||||
// Needs to match what the AArch64 JIT and unaligned signal handler expects
|
||||
uint8_t AtomicFetchNeg(uint8_t *Addr) {
|
||||
using Type = uint8_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrb %w[Result], [%[Memory]];
|
||||
neg %w[Tmp], %w[Result];
|
||||
stlxrb %w[TmpStatus], %w[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint16_t AtomicFetchNeg(uint16_t *Addr) {
|
||||
using Type = uint16_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrh %w[Result], [%[Memory]];
|
||||
neg %w[Tmp], %w[Result];
|
||||
stlxrh %w[TmpStatus], %w[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint32_t AtomicFetchNeg(uint32_t *Addr) {
|
||||
using Type = uint32_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %w[Result], [%[Memory]];
|
||||
neg %w[Tmp], %w[Result];
|
||||
stlxr %w[TmpStatus], %w[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t AtomicFetchNeg(uint64_t *Addr) {
|
||||
using Type = uint64_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %[Result], [%[Memory]];
|
||||
neg %[Tmp], %[Result];
|
||||
stlxr %w[TmpStatus], %[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint8_t AtomicCompareAndSwap(uint8_t expected, uint8_t desired, uint8_t *addr) {
|
||||
using Type = uint8_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrb %w[Tmp], [%[Memory]];
|
||||
cmp %w[Tmp], %w[Expected], uxtb;
|
||||
b.ne 2f;
|
||||
stlxrb %w[Tmp2], %w[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %w[Result], %w[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %w[Result], %w[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint16_t AtomicCompareAndSwap(uint16_t expected, uint16_t desired, uint16_t *addr) {
|
||||
using Type = uint16_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrh %w[Tmp], [%[Memory]];
|
||||
cmp %w[Tmp], %w[Expected], uxth;
|
||||
b.ne 2f;
|
||||
stlxrh %w[Tmp2], %w[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %w[Result], %w[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %w[Result], %w[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint32_t AtomicCompareAndSwap(uint32_t expected, uint32_t desired, uint32_t *addr) {
|
||||
using Type = uint32_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %w[Tmp], [%[Memory]];
|
||||
cmp %w[Tmp], %w[Expected];
|
||||
b.ne 2f;
|
||||
stlxr %w[Tmp2], %w[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %w[Result], %w[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %w[Result], %w[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *addr) {
|
||||
using Type = uint64_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %[Tmp], [%[Memory]];
|
||||
cmp %[Tmp], %[Expected];
|
||||
b.ne 2f;
|
||||
stlxr %w[Tmp2], %[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %[Result], %[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %[Result], %[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Size is the size of each pair element
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Header.Args[2]);
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
__uint128_t Expected = Src1;
|
||||
bool Result = MemData->compare_exchange_strong(Expected, Src2);
|
||||
memcpy(GDP, Result ? &Src1 : &Expected, 16);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CAS) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint8_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint16_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint32_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
|
||||
);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
using Type = uint8_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
using Type = uint16_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
using Type = uint32_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
using Type = uint64_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,170 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->SignalThread(Thread, FEXCore::Core::SignalEvent::Return);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestReturn) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
SignalReturn(Data->State);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
Data->State->CTX->InterpreterCallbackReturn(Data->State, Data->StackEntry);
|
||||
}
|
||||
|
||||
DEF_OP(ExitFunction) {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
|
||||
|
||||
void *ContextData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
memcpy(ContextData, Src, OpSize);
|
||||
|
||||
Data->BlockResults.Quit = true;
|
||||
}
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->Header.Args[0]);
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
bool CompResult;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
else
|
||||
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
|
||||
if (CompResult) {
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TrueBlock);
|
||||
}
|
||||
else {
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->FalseBlock);
|
||||
}
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
|
||||
FEXCore::HLE::SyscallArguments Args;
|
||||
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
|
||||
if (Op->Header.Args[j].IsInvalid()) break;
|
||||
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
|
||||
}
|
||||
|
||||
uint64_t Res = FEXCore::Context::HandleSyscall(Data->State->CTX->SyscallHandler, Data->State->CurrentFrame, &Args);
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
|
||||
FEXCore::HLE::SyscallArguments Args;
|
||||
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
|
||||
if (Op->Header.Args[j].IsInvalid()) break;
|
||||
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
|
||||
}
|
||||
|
||||
// We don't want the errno handling but I also don't want to write inline ASM atm
|
||||
uint64_t Res = syscall(
|
||||
Op->HostSyscallNumber,
|
||||
Args.Argument[0],
|
||||
Args.Argument[1],
|
||||
Args.Argument[2],
|
||||
Args.Argument[3],
|
||||
Args.Argument[4],
|
||||
Args.Argument[5],
|
||||
Args.Argument[6]
|
||||
);
|
||||
|
||||
if (Res == -1) {
|
||||
Res = -errno;
|
||||
}
|
||||
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->Header.Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
|
||||
auto CodePtr = Data->CurrentEntry + Op->Offset;
|
||||
if (memcmp((void*)CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
|
||||
GD = 1;
|
||||
} else {
|
||||
GD = 0;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
Data->State->CTX->RemoveCodeEntry(Data->State, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,224 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
uint64_t Offset = Op->Index * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
Dst |= Src2 << Offset;
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, 8);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, 4);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- float
|
||||
// Only the lower elements from the source
|
||||
// This uses half the source elements
|
||||
uint8_t Elements = OpSize / 8;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(double, float, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
// Little bit tricky here
|
||||
// Sometimes is used to convert from a 128bit vector register
|
||||
// in to a 64bit vector register with different sized elements
|
||||
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
|
||||
uint8_t Elements = (OpSize << 1) / Op->SrcElementSize;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,518 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace AES {
|
||||
static __uint128_t InvShiftRows(uint8_t *State) {
|
||||
uint8_t Shifted[16] = {
|
||||
State[0], State[13], State[10], State[7],
|
||||
State[4], State[1], State[14], State[11],
|
||||
State[8], State[5], State[2], State[15],
|
||||
State[12], State[9], State[6], State[3],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Shifted, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static __uint128_t InvSubBytes(uint8_t *State) {
|
||||
// 16x16 matrix table
|
||||
static const uint8_t InvSubstitutionTable[256] = {
|
||||
0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81, 0xf3, 0xd7, 0xfb,
|
||||
0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e, 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb,
|
||||
0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23, 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e,
|
||||
0x08, 0x2e, 0xa1, 0x66, 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25,
|
||||
0x72, 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65, 0xb6, 0x92,
|
||||
0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46, 0x57, 0xa7, 0x8d, 0x9d, 0x84,
|
||||
0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a, 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06,
|
||||
0xd0, 0x2c, 0x1e, 0x8f, 0xca, 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b,
|
||||
0x3a, 0x91, 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6, 0x73,
|
||||
0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8, 0x1c, 0x75, 0xdf, 0x6e,
|
||||
0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f, 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b,
|
||||
0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2, 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4,
|
||||
0x1f, 0xdd, 0xa8, 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,
|
||||
0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93, 0xc9, 0x9c, 0xef,
|
||||
0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb, 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61,
|
||||
0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6, 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d,
|
||||
};
|
||||
|
||||
// Uses a byte substitution table with a constant set of values
|
||||
// Needs to do a table look up
|
||||
uint8_t Substituted[16];
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
Substituted[i] = InvSubstitutionTable[State[i]];
|
||||
}
|
||||
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Substituted, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static __uint128_t ShiftRows(uint8_t *State) {
|
||||
uint8_t Shifted[16] = {
|
||||
State[0], State[5], State[10], State[15],
|
||||
State[4], State[9], State[14], State[3],
|
||||
State[8], State[13], State[2], State[7],
|
||||
State[12], State[1], State[6], State[11],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Shifted, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static __uint128_t SubBytes(uint8_t *State, size_t Bytes) {
|
||||
// 16x16 matrix table
|
||||
static const uint8_t SubstitutionTable[256] = {
|
||||
0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe, 0xd7, 0xab, 0x76,
|
||||
0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4, 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0,
|
||||
0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7, 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15,
|
||||
0x04, 0xc7, 0x23, 0xc3, 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75,
|
||||
0x09, 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3, 0x2f, 0x84,
|
||||
0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe, 0x39, 0x4a, 0x4c, 0x58, 0xcf,
|
||||
0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85, 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8,
|
||||
0x51, 0xa3, 0x40, 0x8f, 0x92, 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2,
|
||||
0xcd, 0x0c, 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19, 0x73,
|
||||
0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14, 0xde, 0x5e, 0x0b, 0xdb,
|
||||
0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2, 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79,
|
||||
0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5, 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08,
|
||||
0xba, 0x78, 0x25, 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,
|
||||
0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86, 0xc1, 0x1d, 0x9e,
|
||||
0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e, 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf,
|
||||
0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42, 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16,
|
||||
};
|
||||
// Uses a byte substitution table with a constant set of values
|
||||
// Needs to do a table look up
|
||||
uint8_t Substituted[16];
|
||||
Bytes = std::min(Bytes, (size_t)16);
|
||||
for (size_t i = 0; i < Bytes; ++i) {
|
||||
Substituted[i] = SubstitutionTable[State[i]];
|
||||
}
|
||||
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Substituted, Bytes);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static uint8_t FFMul02(uint8_t in) {
|
||||
static const uint8_t FFMul02[256] = {
|
||||
0x00, 0x02, 0x04, 0x06, 0x08, 0x0a, 0x0c, 0x0e, 0x10, 0x12, 0x14, 0x16, 0x18, 0x1a, 0x1c, 0x1e,
|
||||
0x20, 0x22, 0x24, 0x26, 0x28, 0x2a, 0x2c, 0x2e, 0x30, 0x32, 0x34, 0x36, 0x38, 0x3a, 0x3c, 0x3e,
|
||||
0x40, 0x42, 0x44, 0x46, 0x48, 0x4a, 0x4c, 0x4e, 0x50, 0x52, 0x54, 0x56, 0x58, 0x5a, 0x5c, 0x5e,
|
||||
0x60, 0x62, 0x64, 0x66, 0x68, 0x6a, 0x6c, 0x6e, 0x70, 0x72, 0x74, 0x76, 0x78, 0x7a, 0x7c, 0x7e,
|
||||
0x80, 0x82, 0x84, 0x86, 0x88, 0x8a, 0x8c, 0x8e, 0x90, 0x92, 0x94, 0x96, 0x98, 0x9a, 0x9c, 0x9e,
|
||||
0xa0, 0xa2, 0xa4, 0xa6, 0xa8, 0xaa, 0xac, 0xae, 0xb0, 0xb2, 0xb4, 0xb6, 0xb8, 0xba, 0xbc, 0xbe,
|
||||
0xc0, 0xc2, 0xc4, 0xc6, 0xc8, 0xca, 0xcc, 0xce, 0xd0, 0xd2, 0xd4, 0xd6, 0xd8, 0xda, 0xdc, 0xde,
|
||||
0xe0, 0xe2, 0xe4, 0xe6, 0xe8, 0xea, 0xec, 0xee, 0xf0, 0xf2, 0xf4, 0xf6, 0xf8, 0xfa, 0xfc, 0xfe,
|
||||
0x1b, 0x19, 0x1f, 0x1d, 0x13, 0x11, 0x17, 0x15, 0x0b, 0x09, 0x0f, 0x0d, 0x03, 0x01, 0x07, 0x05,
|
||||
0x3b, 0x39, 0x3f, 0x3d, 0x33, 0x31, 0x37, 0x35, 0x2b, 0x29, 0x2f, 0x2d, 0x23, 0x21, 0x27, 0x25,
|
||||
0x5b, 0x59, 0x5f, 0x5d, 0x53, 0x51, 0x57, 0x55, 0x4b, 0x49, 0x4f, 0x4d, 0x43, 0x41, 0x47, 0x45,
|
||||
0x7b, 0x79, 0x7f, 0x7d, 0x73, 0x71, 0x77, 0x75, 0x6b, 0x69, 0x6f, 0x6d, 0x63, 0x61, 0x67, 0x65,
|
||||
0x9b, 0x99, 0x9f, 0x9d, 0x93, 0x91, 0x97, 0x95, 0x8b, 0x89, 0x8f, 0x8d, 0x83, 0x81, 0x87, 0x85,
|
||||
0xbb, 0xb9, 0xbf, 0xbd, 0xb3, 0xb1, 0xb7, 0xb5, 0xab, 0xa9, 0xaf, 0xad, 0xa3, 0xa1, 0xa7, 0xa5,
|
||||
0xdb, 0xd9, 0xdf, 0xdd, 0xd3, 0xd1, 0xd7, 0xd5, 0xcb, 0xc9, 0xcf, 0xcd, 0xc3, 0xc1, 0xc7, 0xc5,
|
||||
0xfb, 0xf9, 0xff, 0xfd, 0xf3, 0xf1, 0xf7, 0xf5, 0xeb, 0xe9, 0xef, 0xed, 0xe3, 0xe1, 0xe7, 0xe5,
|
||||
};
|
||||
return FFMul02[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul03(uint8_t in) {
|
||||
static const uint8_t FFMul03[256] = {
|
||||
0x00, 0x03, 0x06, 0x05, 0x0c, 0x0f, 0x0a, 0x09, 0x18, 0x1b, 0x1e, 0x1d, 0x14, 0x17, 0x12, 0x11,
|
||||
0x30, 0x33, 0x36, 0x35, 0x3c, 0x3f, 0x3a, 0x39, 0x28, 0x2b, 0x2e, 0x2d, 0x24, 0x27, 0x22, 0x21,
|
||||
0x60, 0x63, 0x66, 0x65, 0x6c, 0x6f, 0x6a, 0x69, 0x78, 0x7b, 0x7e, 0x7d, 0x74, 0x77, 0x72, 0x71,
|
||||
0x50, 0x53, 0x56, 0x55, 0x5c, 0x5f, 0x5a, 0x59, 0x48, 0x4b, 0x4e, 0x4d, 0x44, 0x47, 0x42, 0x41,
|
||||
0xc0, 0xc3, 0xc6, 0xc5, 0xcc, 0xcf, 0xca, 0xc9, 0xd8, 0xdb, 0xde, 0xdd, 0xd4, 0xd7, 0xd2, 0xd1,
|
||||
0xf0, 0xf3, 0xf6, 0xf5, 0xfc, 0xff, 0xfa, 0xf9, 0xe8, 0xeb, 0xee, 0xed, 0xe4, 0xe7, 0xe2, 0xe1,
|
||||
0xa0, 0xa3, 0xa6, 0xa5, 0xac, 0xaf, 0xaa, 0xa9, 0xb8, 0xbb, 0xbe, 0xbd, 0xb4, 0xb7, 0xb2, 0xb1,
|
||||
0x90, 0x93, 0x96, 0x95, 0x9c, 0x9f, 0x9a, 0x99, 0x88, 0x8b, 0x8e, 0x8d, 0x84, 0x87, 0x82, 0x81,
|
||||
0x9b, 0x98, 0x9d, 0x9e, 0x97, 0x94, 0x91, 0x92, 0x83, 0x80, 0x85, 0x86, 0x8f, 0x8c, 0x89, 0x8a,
|
||||
0xab, 0xa8, 0xad, 0xae, 0xa7, 0xa4, 0xa1, 0xa2, 0xb3, 0xb0, 0xb5, 0xb6, 0xbf, 0xbc, 0xb9, 0xba,
|
||||
0xfb, 0xf8, 0xfd, 0xfe, 0xf7, 0xf4, 0xf1, 0xf2, 0xe3, 0xe0, 0xe5, 0xe6, 0xef, 0xec, 0xe9, 0xea,
|
||||
0xcb, 0xc8, 0xcd, 0xce, 0xc7, 0xc4, 0xc1, 0xc2, 0xd3, 0xd0, 0xd5, 0xd6, 0xdf, 0xdc, 0xd9, 0xda,
|
||||
0x5b, 0x58, 0x5d, 0x5e, 0x57, 0x54, 0x51, 0x52, 0x43, 0x40, 0x45, 0x46, 0x4f, 0x4c, 0x49, 0x4a,
|
||||
0x6b, 0x68, 0x6d, 0x6e, 0x67, 0x64, 0x61, 0x62, 0x73, 0x70, 0x75, 0x76, 0x7f, 0x7c, 0x79, 0x7a,
|
||||
0x3b, 0x38, 0x3d, 0x3e, 0x37, 0x34, 0x31, 0x32, 0x23, 0x20, 0x25, 0x26, 0x2f, 0x2c, 0x29, 0x2a,
|
||||
0x0b, 0x08, 0x0d, 0x0e, 0x07, 0x04, 0x01, 0x02, 0x13, 0x10, 0x15, 0x16, 0x1f, 0x1c, 0x19, 0x1a,
|
||||
};
|
||||
return FFMul03[in];
|
||||
}
|
||||
|
||||
static __uint128_t MixColumns(uint8_t *State) {
|
||||
uint8_t In0[16] = {
|
||||
State[0], State[4], State[8], State[12],
|
||||
State[1], State[5], State[9], State[13],
|
||||
State[2], State[6], State[10], State[14],
|
||||
State[3], State[7], State[11], State[15],
|
||||
};
|
||||
|
||||
uint8_t Out0[4]{};
|
||||
uint8_t Out1[4]{};
|
||||
uint8_t Out2[4]{};
|
||||
uint8_t Out3[4]{};
|
||||
|
||||
for (size_t i = 0; i < 4; ++i) {
|
||||
Out0[i] = FFMul02(In0[0 + i]) ^ FFMul03(In0[4 + i]) ^ In0[8 + i] ^ In0[12 + i];
|
||||
Out1[i] = In0[0 + i] ^ FFMul02(In0[4 + i]) ^ FFMul03(In0[8 + i]) ^ In0[12 + i];
|
||||
Out2[i] = In0[0 + i] ^ In0[4 + i] ^ FFMul02(In0[8 + i]) ^ FFMul03(In0[12 + i]);
|
||||
Out3[i] = FFMul03(In0[0 + i]) ^ In0[4 + i] ^ In0[8 + i] ^ FFMul02(In0[12 + i]);
|
||||
}
|
||||
|
||||
uint8_t OutArray[16] = {
|
||||
Out0[0], Out1[0], Out2[0], Out3[0],
|
||||
Out0[1], Out1[1], Out2[1], Out3[1],
|
||||
Out0[2], Out1[2], Out2[2], Out3[2],
|
||||
Out0[3], Out1[3], Out2[3], Out3[3],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, OutArray, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static uint8_t FFMul09(uint8_t in) {
|
||||
static const uint8_t FFMul09[256] = {
|
||||
0x00, 0x09, 0x12, 0x1b, 0x24, 0x2d, 0x36, 0x3f, 0x48, 0x41, 0x5a, 0x53, 0x6c, 0x65, 0x7e, 0x77,
|
||||
0x90, 0x99, 0x82, 0x8b, 0xb4, 0xbd, 0xa6, 0xaf, 0xd8, 0xd1, 0xca, 0xc3, 0xfc, 0xf5, 0xee, 0xe7,
|
||||
0x3b, 0x32, 0x29, 0x20, 0x1f, 0x16, 0x0d, 0x04, 0x73, 0x7a, 0x61, 0x68, 0x57, 0x5e, 0x45, 0x4c,
|
||||
0xab, 0xa2, 0xb9, 0xb0, 0x8f, 0x86, 0x9d, 0x94, 0xe3, 0xea, 0xf1, 0xf8, 0xc7, 0xce, 0xd5, 0xdc,
|
||||
0x76, 0x7f, 0x64, 0x6d, 0x52, 0x5b, 0x40, 0x49, 0x3e, 0x37, 0x2c, 0x25, 0x1a, 0x13, 0x08, 0x01,
|
||||
0xe6, 0xef, 0xf4, 0xfd, 0xc2, 0xcb, 0xd0, 0xd9, 0xae, 0xa7, 0xbc, 0xb5, 0x8a, 0x83, 0x98, 0x91,
|
||||
0x4d, 0x44, 0x5f, 0x56, 0x69, 0x60, 0x7b, 0x72, 0x05, 0x0c, 0x17, 0x1e, 0x21, 0x28, 0x33, 0x3a,
|
||||
0xdd, 0xd4, 0xcf, 0xc6, 0xf9, 0xf0, 0xeb, 0xe2, 0x95, 0x9c, 0x87, 0x8e, 0xb1, 0xb8, 0xa3, 0xaa,
|
||||
0xec, 0xe5, 0xfe, 0xf7, 0xc8, 0xc1, 0xda, 0xd3, 0xa4, 0xad, 0xb6, 0xbf, 0x80, 0x89, 0x92, 0x9b,
|
||||
0x7c, 0x75, 0x6e, 0x67, 0x58, 0x51, 0x4a, 0x43, 0x34, 0x3d, 0x26, 0x2f, 0x10, 0x19, 0x02, 0x0b,
|
||||
0xd7, 0xde, 0xc5, 0xcc, 0xf3, 0xfa, 0xe1, 0xe8, 0x9f, 0x96, 0x8d, 0x84, 0xbb, 0xb2, 0xa9, 0xa0,
|
||||
0x47, 0x4e, 0x55, 0x5c, 0x63, 0x6a, 0x71, 0x78, 0x0f, 0x06, 0x1d, 0x14, 0x2b, 0x22, 0x39, 0x30,
|
||||
0x9a, 0x93, 0x88, 0x81, 0xbe, 0xb7, 0xac, 0xa5, 0xd2, 0xdb, 0xc0, 0xc9, 0xf6, 0xff, 0xe4, 0xed,
|
||||
0x0a, 0x03, 0x18, 0x11, 0x2e, 0x27, 0x3c, 0x35, 0x42, 0x4b, 0x50, 0x59, 0x66, 0x6f, 0x74, 0x7d,
|
||||
0xa1, 0xa8, 0xb3, 0xba, 0x85, 0x8c, 0x97, 0x9e, 0xe9, 0xe0, 0xfb, 0xf2, 0xcd, 0xc4, 0xdf, 0xd6,
|
||||
0x31, 0x38, 0x23, 0x2a, 0x15, 0x1c, 0x07, 0x0e, 0x79, 0x70, 0x6b, 0x62, 0x5d, 0x54, 0x4f, 0x46,
|
||||
};
|
||||
return FFMul09[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul0B(uint8_t in) {
|
||||
static const uint8_t FFMul0B[256] = {
|
||||
0x00, 0x0b, 0x16, 0x1d, 0x2c, 0x27, 0x3a, 0x31, 0x58, 0x53, 0x4e, 0x45, 0x74, 0x7f, 0x62, 0x69,
|
||||
0xb0, 0xbb, 0xa6, 0xad, 0x9c, 0x97, 0x8a, 0x81, 0xe8, 0xe3, 0xfe, 0xf5, 0xc4, 0xcf, 0xd2, 0xd9,
|
||||
0x7b, 0x70, 0x6d, 0x66, 0x57, 0x5c, 0x41, 0x4a, 0x23, 0x28, 0x35, 0x3e, 0x0f, 0x04, 0x19, 0x12,
|
||||
0xcb, 0xc0, 0xdd, 0xd6, 0xe7, 0xec, 0xf1, 0xfa, 0x93, 0x98, 0x85, 0x8e, 0xbf, 0xb4, 0xa9, 0xa2,
|
||||
0xf6, 0xfd, 0xe0, 0xeb, 0xda, 0xd1, 0xcc, 0xc7, 0xae, 0xa5, 0xb8, 0xb3, 0x82, 0x89, 0x94, 0x9f,
|
||||
0x46, 0x4d, 0x50, 0x5b, 0x6a, 0x61, 0x7c, 0x77, 0x1e, 0x15, 0x08, 0x03, 0x32, 0x39, 0x24, 0x2f,
|
||||
0x8d, 0x86, 0x9b, 0x90, 0xa1, 0xaa, 0xb7, 0xbc, 0xd5, 0xde, 0xc3, 0xc8, 0xf9, 0xf2, 0xef, 0xe4,
|
||||
0x3d, 0x36, 0x2b, 0x20, 0x11, 0x1a, 0x07, 0x0c, 0x65, 0x6e, 0x73, 0x78, 0x49, 0x42, 0x5f, 0x54,
|
||||
0xf7, 0xfc, 0xe1, 0xea, 0xdb, 0xd0, 0xcd, 0xc6, 0xaf, 0xa4, 0xb9, 0xb2, 0x83, 0x88, 0x95, 0x9e,
|
||||
0x47, 0x4c, 0x51, 0x5a, 0x6b, 0x60, 0x7d, 0x76, 0x1f, 0x14, 0x09, 0x02, 0x33, 0x38, 0x25, 0x2e,
|
||||
0x8c, 0x87, 0x9a, 0x91, 0xa0, 0xab, 0xb6, 0xbd, 0xd4, 0xdf, 0xc2, 0xc9, 0xf8, 0xf3, 0xee, 0xe5,
|
||||
0x3c, 0x37, 0x2a, 0x21, 0x10, 0x1b, 0x06, 0x0d, 0x64, 0x6f, 0x72, 0x79, 0x48, 0x43, 0x5e, 0x55,
|
||||
0x01, 0x0a, 0x17, 0x1c, 0x2d, 0x26, 0x3b, 0x30, 0x59, 0x52, 0x4f, 0x44, 0x75, 0x7e, 0x63, 0x68,
|
||||
0xb1, 0xba, 0xa7, 0xac, 0x9d, 0x96, 0x8b, 0x80, 0xe9, 0xe2, 0xff, 0xf4, 0xc5, 0xce, 0xd3, 0xd8,
|
||||
0x7a, 0x71, 0x6c, 0x67, 0x56, 0x5d, 0x40, 0x4b, 0x22, 0x29, 0x34, 0x3f, 0x0e, 0x05, 0x18, 0x13,
|
||||
0xca, 0xc1, 0xdc, 0xd7, 0xe6, 0xed, 0xf0, 0xfb, 0x92, 0x99, 0x84, 0x8f, 0xbe, 0xb5, 0xa8, 0xa3,
|
||||
};
|
||||
return FFMul0B[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul0D(uint8_t in) {
|
||||
static const uint8_t FFMul0D[256] = {
|
||||
0x00, 0x0d, 0x1a, 0x17, 0x34, 0x39, 0x2e, 0x23, 0x68, 0x65, 0x72, 0x7f, 0x5c, 0x51, 0x46, 0x4b,
|
||||
0xd0, 0xdd, 0xca, 0xc7, 0xe4, 0xe9, 0xfe, 0xf3, 0xb8, 0xb5, 0xa2, 0xaf, 0x8c, 0x81, 0x96, 0x9b,
|
||||
0xbb, 0xb6, 0xa1, 0xac, 0x8f, 0x82, 0x95, 0x98, 0xd3, 0xde, 0xc9, 0xc4, 0xe7, 0xea, 0xfd, 0xf0,
|
||||
0x6b, 0x66, 0x71, 0x7c, 0x5f, 0x52, 0x45, 0x48, 0x03, 0x0e, 0x19, 0x14, 0x37, 0x3a, 0x2d, 0x20,
|
||||
0x6d, 0x60, 0x77, 0x7a, 0x59, 0x54, 0x43, 0x4e, 0x05, 0x08, 0x1f, 0x12, 0x31, 0x3c, 0x2b, 0x26,
|
||||
0xbd, 0xb0, 0xa7, 0xaa, 0x89, 0x84, 0x93, 0x9e, 0xd5, 0xd8, 0xcf, 0xc2, 0xe1, 0xec, 0xfb, 0xf6,
|
||||
0xd6, 0xdb, 0xcc, 0xc1, 0xe2, 0xef, 0xf8, 0xf5, 0xbe, 0xb3, 0xa4, 0xa9, 0x8a, 0x87, 0x90, 0x9d,
|
||||
0x06, 0x0b, 0x1c, 0x11, 0x32, 0x3f, 0x28, 0x25, 0x6e, 0x63, 0x74, 0x79, 0x5a, 0x57, 0x40, 0x4d,
|
||||
0xda, 0xd7, 0xc0, 0xcd, 0xee, 0xe3, 0xf4, 0xf9, 0xb2, 0xbf, 0xa8, 0xa5, 0x86, 0x8b, 0x9c, 0x91,
|
||||
0x0a, 0x07, 0x10, 0x1d, 0x3e, 0x33, 0x24, 0x29, 0x62, 0x6f, 0x78, 0x75, 0x56, 0x5b, 0x4c, 0x41,
|
||||
0x61, 0x6c, 0x7b, 0x76, 0x55, 0x58, 0x4f, 0x42, 0x09, 0x04, 0x13, 0x1e, 0x3d, 0x30, 0x27, 0x2a,
|
||||
0xb1, 0xbc, 0xab, 0xa6, 0x85, 0x88, 0x9f, 0x92, 0xd9, 0xd4, 0xc3, 0xce, 0xed, 0xe0, 0xf7, 0xfa,
|
||||
0xb7, 0xba, 0xad, 0xa0, 0x83, 0x8e, 0x99, 0x94, 0xdf, 0xd2, 0xc5, 0xc8, 0xeb, 0xe6, 0xf1, 0xfc,
|
||||
0x67, 0x6a, 0x7d, 0x70, 0x53, 0x5e, 0x49, 0x44, 0x0f, 0x02, 0x15, 0x18, 0x3b, 0x36, 0x21, 0x2c,
|
||||
0x0c, 0x01, 0x16, 0x1b, 0x38, 0x35, 0x22, 0x2f, 0x64, 0x69, 0x7e, 0x73, 0x50, 0x5d, 0x4a, 0x47,
|
||||
0xdc, 0xd1, 0xc6, 0xcb, 0xe8, 0xe5, 0xf2, 0xff, 0xb4, 0xb9, 0xae, 0xa3, 0x80, 0x8d, 0x9a, 0x97,
|
||||
};
|
||||
|
||||
return FFMul0D[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul0E(uint8_t in) {
|
||||
static const uint8_t FFMul0E[256] = {
|
||||
0x00, 0x0e, 0x1c, 0x12, 0x38, 0x36, 0x24, 0x2a, 0x70, 0x7e, 0x6c, 0x62, 0x48, 0x46, 0x54, 0x5a,
|
||||
0xe0, 0xee, 0xfc, 0xf2, 0xd8, 0xd6, 0xc4, 0xca, 0x90, 0x9e, 0x8c, 0x82, 0xa8, 0xa6, 0xb4, 0xba,
|
||||
0xdb, 0xd5, 0xc7, 0xc9, 0xe3, 0xed, 0xff, 0xf1, 0xab, 0xa5, 0xb7, 0xb9, 0x93, 0x9d, 0x8f, 0x81,
|
||||
0x3b, 0x35, 0x27, 0x29, 0x03, 0x0d, 0x1f, 0x11, 0x4b, 0x45, 0x57, 0x59, 0x73, 0x7d, 0x6f, 0x61,
|
||||
0xad, 0xa3, 0xb1, 0xbf, 0x95, 0x9b, 0x89, 0x87, 0xdd, 0xd3, 0xc1, 0xcf, 0xe5, 0xeb, 0xf9, 0xf7,
|
||||
0x4d, 0x43, 0x51, 0x5f, 0x75, 0x7b, 0x69, 0x67, 0x3d, 0x33, 0x21, 0x2f, 0x05, 0x0b, 0x19, 0x17,
|
||||
0x76, 0x78, 0x6a, 0x64, 0x4e, 0x40, 0x52, 0x5c, 0x06, 0x08, 0x1a, 0x14, 0x3e, 0x30, 0x22, 0x2c,
|
||||
0x96, 0x98, 0x8a, 0x84, 0xae, 0xa0, 0xb2, 0xbc, 0xe6, 0xe8, 0xfa, 0xf4, 0xde, 0xd0, 0xc2, 0xcc,
|
||||
0x41, 0x4f, 0x5d, 0x53, 0x79, 0x77, 0x65, 0x6b, 0x31, 0x3f, 0x2d, 0x23, 0x09, 0x07, 0x15, 0x1b,
|
||||
0xa1, 0xaf, 0xbd, 0xb3, 0x99, 0x97, 0x85, 0x8b, 0xd1, 0xdf, 0xcd, 0xc3, 0xe9, 0xe7, 0xf5, 0xfb,
|
||||
0x9a, 0x94, 0x86, 0x88, 0xa2, 0xac, 0xbe, 0xb0, 0xea, 0xe4, 0xf6, 0xf8, 0xd2, 0xdc, 0xce, 0xc0,
|
||||
0x7a, 0x74, 0x66, 0x68, 0x42, 0x4c, 0x5e, 0x50, 0x0a, 0x04, 0x16, 0x18, 0x32, 0x3c, 0x2e, 0x20,
|
||||
0xec, 0xe2, 0xf0, 0xfe, 0xd4, 0xda, 0xc8, 0xc6, 0x9c, 0x92, 0x80, 0x8e, 0xa4, 0xaa, 0xb8, 0xb6,
|
||||
0x0c, 0x02, 0x10, 0x1e, 0x34, 0x3a, 0x28, 0x26, 0x7c, 0x72, 0x60, 0x6e, 0x44, 0x4a, 0x58, 0x56,
|
||||
0x37, 0x39, 0x2b, 0x25, 0x0f, 0x01, 0x13, 0x1d, 0x47, 0x49, 0x5b, 0x55, 0x7f, 0x71, 0x63, 0x6d,
|
||||
0xd7, 0xd9, 0xcb, 0xc5, 0xef, 0xe1, 0xf3, 0xfd, 0xa7, 0xa9, 0xbb, 0xb5, 0x9f, 0x91, 0x83, 0x8d,
|
||||
};
|
||||
|
||||
return FFMul0E[in];
|
||||
}
|
||||
|
||||
static __uint128_t InvMixColumns(uint8_t *State) {
|
||||
uint8_t In0[16] = {
|
||||
State[0], State[4], State[8], State[12],
|
||||
State[1], State[5], State[9], State[13],
|
||||
State[2], State[6], State[10], State[14],
|
||||
State[3], State[7], State[11], State[15],
|
||||
};
|
||||
|
||||
uint8_t Out0[4]{};
|
||||
uint8_t Out1[4]{};
|
||||
uint8_t Out2[4]{};
|
||||
uint8_t Out3[4]{};
|
||||
|
||||
for (size_t i = 0; i < 4; ++i) {
|
||||
Out0[i] = FFMul0E(In0[0 + i]) ^ FFMul0B(In0[4 + i]) ^ FFMul0D(In0[8 + i]) ^ FFMul09(In0[12 + i]);
|
||||
Out1[i] = FFMul09(In0[0 + i]) ^ FFMul0E(In0[4 + i]) ^ FFMul0B(In0[8 + i]) ^ FFMul0D(In0[12 + i]);
|
||||
Out2[i] = FFMul0D(In0[0 + i]) ^ FFMul09(In0[4 + i]) ^ FFMul0E(In0[8 + i]) ^ FFMul0B(In0[12 + i]);
|
||||
Out3[i] = FFMul0B(In0[0 + i]) ^ FFMul0D(In0[4 + i]) ^ FFMul09(In0[8 + i]) ^ FFMul0E(In0[12 + i]);
|
||||
}
|
||||
|
||||
uint8_t OutArray[16] = {
|
||||
Out0[0], Out1[0], Out2[0], Out3[0],
|
||||
Out0[1], Out1[1], Out2[1], Out3[1],
|
||||
Out0[2], Out1[2], Out2[2], Out3[2],
|
||||
Out0[3], Out1[3], Out2[3], Out3[3],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, OutArray, 16);
|
||||
return Res;
|
||||
}
|
||||
}
|
||||
|
||||
namespace CRC32 {
|
||||
// CRC32 per byte lookup table.
|
||||
constexpr std::array<uint32_t, 256> CRC32CTable = []() consteval {
|
||||
std::array<uint32_t, 256> Table{};
|
||||
|
||||
// Clang 11.x doesn't support bitreverse as a consteval
|
||||
// constexpr uint32_t Polynomial = 0x1EDC6F41;
|
||||
constexpr uint32_t PolynomialRev = 0x82F63B78; //__builtin_bitreverse32(Polynomial);
|
||||
|
||||
for (size_t Char = 0; Char < std::size(Table); ++Char) {
|
||||
uint32_t CurrentChar = Char;
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
if (CurrentChar & 1) {
|
||||
CurrentChar = (CurrentChar >> 1) ^ PolynomialRev;
|
||||
}
|
||||
else {
|
||||
CurrentChar >>= 1;
|
||||
}
|
||||
}
|
||||
Table[Char] = CurrentChar;
|
||||
}
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
uint32_t crc32cb(uint32_t Accumulator, uint8_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ data] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32ch(uint32_t Accumulator, uint16_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cw(uint32_t Accumulator, uint32_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cx(uint32_t Accumulator, uint64_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 32) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 40) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 48) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 56) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
// Pseudo-code
|
||||
// Dst = InvMixColumns(STATE)
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Src1));
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = ShiftRows(STATE)
|
||||
// STATE = SubBytes(STATE)
|
||||
// STATE = MixColumns(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
|
||||
Tmp = AES::MixColumns(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = ShiftRows(STATE)
|
||||
// STATE = SubBytes(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = InvShiftRows(STATE)
|
||||
// STATE = InvSubBytes(STATE)
|
||||
// STATE = InvMixColumns(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = InvShiftRows(STATE)
|
||||
// STATE = InvSubBytes(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
// Pseudo-code
|
||||
// X3 = Src1[127:96]
|
||||
// X2 = Src1[95:64]
|
||||
// X1 = Src1[63:32]
|
||||
// X0 = Src1[31:30]
|
||||
// RCON = (Zext)rcon
|
||||
// Dest[31:0] = SubWord(X1)
|
||||
// Dest[63:32] = RotWord(SubWord(X1)) XOR RCON
|
||||
// Dest[95:64] = SubWord(X3)
|
||||
// Dest[127:96] = RotWord(SubWord(X3)) XOR RCON
|
||||
__uint128_t Tmp{};
|
||||
uint32_t X1{};
|
||||
uint32_t X3{};
|
||||
memcpy(&X1, &Src1[4], 4);
|
||||
memcpy(&X3, &Src1[12], 4);
|
||||
uint32_t SubWord_X1 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X1), 4);
|
||||
uint32_t SubWord_X3 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X3), 4);
|
||||
|
||||
auto Ror = [] (auto In, auto R) {
|
||||
auto RotateMask = sizeof(In) * 8 - 1;
|
||||
R &= RotateMask;
|
||||
return (In >> R) | (In << (sizeof(In) * 8 - R));
|
||||
};
|
||||
|
||||
uint32_t Rot_X1 = Ror(SubWord_X1, 8);
|
||||
uint32_t Rot_X3 = Ror(SubWord_X3, 8);
|
||||
|
||||
Tmp = Rot_X3 ^ Op->RCON;
|
||||
Tmp <<= 32;
|
||||
Tmp |= SubWord_X3;
|
||||
Tmp <<= 32;
|
||||
Tmp |= Rot_X1 ^ Op->RCON;
|
||||
Tmp <<= 32;
|
||||
Tmp |= SubWord_X1;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
uint32_t Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Src1);
|
||||
uint8_t *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->Src2);
|
||||
uint32_t Tmp{};
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
Tmp = CRC32::crc32cb(Src1, *(uint8_t*)Src2);
|
||||
break;
|
||||
case 2:
|
||||
Tmp = CRC32::crc32ch(Src1, *(uint16_t*)Src2);
|
||||
break;
|
||||
case 4:
|
||||
Tmp = CRC32::crc32cw(Src1, *(uint32_t*)Src2);
|
||||
break;
|
||||
case 8:
|
||||
Tmp = CRC32::crc32cx(Src1, *(uint64_t*)Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown CRC32C size: {}", Op->SrcSize);
|
||||
break;
|
||||
|
||||
}
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,361 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include "F80Ops.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(F80LOADFCW) {
|
||||
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(F80ADD) {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SUB) {
|
||||
auto Op = IROp->C<IR::IROp_F80Sub>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80MUL) {
|
||||
auto Op = IROp->C<IR::IROp_F80Mul>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80DIV) {
|
||||
auto Op = IROp->C<IR::IROp_F80Div>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F80FYL2X>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80ATAN>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM1>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F80SCALE>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CVT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVT>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
float Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTTO) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 4: {
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTTOINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 2: {
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80ROUND) {
|
||||
auto Op = IROp->C<IR::IROp_F80Round>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80F2XM1>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::F2XM1(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80TAN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FTAN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SQRT) {
|
||||
auto Op = IROp->C<IR::IROp_F80SQRT>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSQRT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F80SIN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSIN(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80COS) {
|
||||
auto Op = IROp->C<IR::IROp_F80COS>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FCOS(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_EXP) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_SIG) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CMP) {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
uint32_t ResultFlags{};
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool eq, lt, nan;
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
|
||||
GD = ResultFlags;
|
||||
}
|
||||
|
||||
DEF_OP(F80BCDLOAD) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
// | 4 bit | 4 bit |
|
||||
// | 10s place | 1s place |
|
||||
// EG 0x48 = 48
|
||||
// EG 0x4847 = 4847
|
||||
// This gives us an 18digit value encoded in BCD
|
||||
// The last byte lets us know if it negative or not
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
uint8_t Digit = Src1[8 - i];
|
||||
// First shift our last value over
|
||||
BCD *= 100;
|
||||
|
||||
// Add the tens place digit
|
||||
BCD += (Digit >> 4) * 10;
|
||||
|
||||
// Add the ones place digit
|
||||
BCD += Digit & 0xF;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
bool Negative = Src1[9] & 0x80;
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Sign = Negative;
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
uint8_t BCD[10]{};
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
// Nothing left? Just leave
|
||||
break;
|
||||
}
|
||||
// Extract the lower 100 values
|
||||
uint8_t Digit = Tmp % 100;
|
||||
|
||||
// Now divide it for the next iteration
|
||||
Tmp /= 100;
|
||||
|
||||
uint8_t UpperNibble = Digit / 10;
|
||||
uint8_t LowerNibble = Digit % 10;
|
||||
|
||||
// Now store the BCD
|
||||
BCD[i] = (UpperNibble << 4) | LowerNibble;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
BCD[9] = Negative ? 0x80 : 0;
|
||||
|
||||
memcpy(GDP, BCD, 10);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,332 @@
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/SoftFloat-3e/softfloat.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
template<IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
static X80SoftFloat handle4(float src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle8(double src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
template<uint32_t Flags>
|
||||
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
bool eq, lt, nan;
|
||||
uint64_t ResultFlags = 0;
|
||||
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
return ResultFlags;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
static float handle4(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static double handle8(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
static int16_t handle2(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int32_t handle4(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int64_t handle8(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int16_t handle2t(X80SoftFloat src) {
|
||||
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX) {
|
||||
return INT16_MAX;
|
||||
} else if (rv < INT16_MIN) {
|
||||
return INT16_MIN;
|
||||
} else {
|
||||
return rv;
|
||||
}
|
||||
}
|
||||
|
||||
static int32_t handle4t(X80SoftFloat src) {
|
||||
return extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
static int64_t handle8t(X80SoftFloat src) {
|
||||
return extF80_to_i64(src, softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
static X80SoftFloat handle2(int16_t src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle4(int32_t src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FRNDINT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::F2XM1(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FTAN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FSQRT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FSIN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FCOS(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FADD(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FSUB(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FMUL(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FDIV(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FYL2X(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FATAN(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FREM1(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FREM(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FSCALE(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(Src1);
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
X80SoftFloat Rv;
|
||||
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
// Nothing left? Just leave
|
||||
break;
|
||||
}
|
||||
// Extract the lower 100 values
|
||||
uint8_t Digit = Tmp % 100;
|
||||
|
||||
// Now divide it for the next iteration
|
||||
Tmp /= 100;
|
||||
|
||||
uint8_t UpperNibble = Digit / 10;
|
||||
uint8_t LowerNibble = Digit % 10;
|
||||
|
||||
// Now store the BCD
|
||||
BCD[i] = (UpperNibble << 4) | LowerNibble;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
BCD[9] = Negative ? 0x80 : 0;
|
||||
|
||||
return Rv;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src) {
|
||||
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
// | 4 bit | 4 bit |
|
||||
// | 10s place | 1s place |
|
||||
// EG 0x48 = 48
|
||||
// EG 0x4847 = 4847
|
||||
// This gives us an 18digit value encoded in BCD
|
||||
// The last byte lets us know if it negative or not
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
uint8_t Digit = Src1[8 - i];
|
||||
// First shift our last value over
|
||||
BCD *= 100;
|
||||
|
||||
// Add the tens place digit
|
||||
BCD += (Digit >> 4) * 10;
|
||||
|
||||
// Add the ones place digit
|
||||
BCD += Digit & 0xF;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
bool Negative = Src1[9] & 0x80;
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Sign = Negative;
|
||||
return Tmp;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80LOADFCW> {
|
||||
static void handle(uint16_t NewFCW) {
|
||||
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
switch(PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
switch(RC) {
|
||||
case 0:
|
||||
softfloat_roundingMode = softfloat_round_near_even;
|
||||
break;
|
||||
case 1:
|
||||
softfloat_roundingMode = softfloat_round_min;
|
||||
break;
|
||||
case 2:
|
||||
softfloat_roundingMode = softfloat_round_max;
|
||||
break;
|
||||
case 3:
|
||||
softfloat_roundingMode = softfloat_round_minMag;
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]) >> Op->Flag) & 1;
|
||||
}
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -20,31 +20,36 @@ using DestMapType = std::vector<uint32_t>;
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
std::string GetName() override { return "Interpreter"; }
|
||||
void *CompileCode(uint64_t Entry, FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
|
||||
bool NeedsOpDispatch() override { return true; }
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *State;
|
||||
|
||||
uint32_t AllocateTmpSpace(size_t Size);
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, IR::OrderedNodeWrapper Op);
|
||||
|
||||
template<typename Res>
|
||||
Res GetSrc(void* SSAData, IR::OrderedNodeWrapper Src);
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher{};
|
||||
};
|
||||
|
||||
}
|
||||
template<typename T>
|
||||
T AtomicCompareAndSwap(T expected, T desired, T *addr);
|
||||
|
||||
uint8_t AtomicFetchNeg(uint8_t *Addr);
|
||||
uint16_t AtomicFetchNeg(uint16_t *Addr);
|
||||
uint32_t AtomicFetchNeg(uint32_t *Addr);
|
||||
uint64_t AtomicFetchNeg(uint64_t *Addr);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,30 +1,31 @@
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/DebugData.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <atomic>
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
#include <memory>
|
||||
#include <bits/types/stack_t.h>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
|
||||
#include "InterpreterOps.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
@@ -34,70 +35,9 @@ static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
InterpreterOps::InterpretIR(Thread, Thread->CurrentFrame->State.rip, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
}
|
||||
|
||||
bool InterpreterCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
#ifdef _M_ARM_64
|
||||
constexpr bool is_arm64 = true;
|
||||
#else
|
||||
constexpr bool is_arm64 = false;
|
||||
#endif
|
||||
|
||||
if constexpr (is_arm64) {
|
||||
uint32_t *PC = reinterpret_cast<uint32_t*>(ArchHelpers::Context::GetPc(ucontext));
|
||||
uint32_t Instr = PC[0];
|
||||
if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXR_MASK) == FEXCore::ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ucontext, info);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + BytesToSkip);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS LDAXR: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: CTX {ctx}
|
||||
, State {Thread} {
|
||||
// Grab our space for temporary data
|
||||
|
||||
if (!CompileThread &&
|
||||
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
@@ -107,17 +47,18 @@ InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->HandleSIGBUS(Signal, info, ucontext);
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,8 @@ namespace FEXCore::Core {
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,179 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#define GD *GetDest<uint64_t*>(Data->SSAData, Node)
|
||||
#define GDP GetDest<void*>(Data->SSAData, Node)
|
||||
|
||||
#define DO_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(GDP); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
*Dst_d = func(*Src1_d, *Src2_d); \
|
||||
break; \
|
||||
}
|
||||
#define DO_SCALAR_COMPARE_OP(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
Dst_d[0] = func(Src1_d[0], Src2_d[0]); \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_COMPARE_OP(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_PAIR_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i*2], Src1_d[i*2 + 1]); \
|
||||
Dst_d[i+Elements] = func(Src2_d[i*2], Src2_d[i*2 + 1]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_SCALAR_OP(size, type, func)\
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], *Src2_d); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_0SRC_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_1SRC_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_REDUCE_1SRC_OP(size, type, func, start_val) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type*>(Src); \
|
||||
type begin = start_val; \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
begin = func(begin, Src_d[i]); \
|
||||
} \
|
||||
Dst_d[0] = begin; \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_SAT_OP(size, type, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], Src2_d[i], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP(size, type, type2, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func(Src_d[i], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(type, type2, func, min, max) \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func(Src_d[i], min, max); \
|
||||
}
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP_TOP(size, type, type2, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src2); \
|
||||
memcpy(Dst_d, Src1, Elements * sizeof(type2));\
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(size, type, type2, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func(Src_d[i+Elements], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_2SRC_2TYPE_OP(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func((type)Src1_d[i], (type)Src2_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func((type)Src1_d[i+Elements], (type)Src2_d[i+Elements]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
|
||||
template<typename Res>
|
||||
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
|
||||
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Src.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/F80Ops.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...)) {
|
||||
return {FABI_UNKNOWN, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float)) {
|
||||
return {FABI_F80_F32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double)) {
|
||||
return {FABI_F80_F64, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t)) {
|
||||
return {FABI_F80_I16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t)) {
|
||||
return {FABI_VOID_U16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t)) {
|
||||
return {FABI_F80_I32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I16_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_I64_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_F80_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags]);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
+325
-5403
File diff suppressed because it is too large.
Load diff
@@ -1,9 +1,16 @@
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
struct IROp_Header;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core{
|
||||
@@ -32,11 +39,363 @@ namespace FEXCore::CPU {
|
||||
FallbackABI ABI;
|
||||
void *fn;
|
||||
};
|
||||
|
||||
|
||||
class InterpreterOps {
|
||||
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
uint64_t CurrentEntry{};
|
||||
FEXCore::IR::IRListView *CurrentIR{};
|
||||
volatile void *StackEntry{};
|
||||
void *SSAData{};
|
||||
struct {
|
||||
bool Quit;
|
||||
bool Redo;
|
||||
} BlockResults{};
|
||||
|
||||
IR::NodeIterator BlockIterator{0, 0};
|
||||
};
|
||||
|
||||
#define DEF_OP(x) static void Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
///< No-op Handler
|
||||
DEF_OP(NoOp);
|
||||
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
DEF_OP(UDiv);
|
||||
DEF_OP(Rem);
|
||||
DEF_OP(URem);
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
DEF_OP(Ashr);
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
DEF_OP(LURem);
|
||||
DEF_OP(Zext);
|
||||
DEF_OP(Not);
|
||||
DEF_OP(Popcount);
|
||||
DEF_OP(FindLSB);
|
||||
DEF_OP(FindMSB);
|
||||
DEF_OP(FindTrailingZeros);
|
||||
DEF_OP(CountLeadingZeroes);
|
||||
DEF_OP(Rev);
|
||||
DEF_OP(Bfi);
|
||||
DEF_OP(Bfe);
|
||||
DEF_OP(Sbfe);
|
||||
DEF_OP(Select);
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
///< Atomic ops
|
||||
DEF_OP(CASPair);
|
||||
DEF_OP(CAS);
|
||||
DEF_OP(AtomicAdd);
|
||||
DEF_OP(AtomicSub);
|
||||
DEF_OP(AtomicAnd);
|
||||
DEF_OP(AtomicOr);
|
||||
DEF_OP(AtomicXor);
|
||||
DEF_OP(AtomicSwap);
|
||||
DEF_OP(AtomicFetchAdd);
|
||||
DEF_OP(AtomicFetchSub);
|
||||
DEF_OP(AtomicFetchAnd);
|
||||
DEF_OP(AtomicFetchOr);
|
||||
DEF_OP(AtomicFetchXor);
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(GuestReturn);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
DEF_OP(FillRegister);
|
||||
DEF_OP(LoadFlag);
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
DEF_OP(PhiValue);
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(CreateVector2);
|
||||
DEF_OP(CreateVector4);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
DEF_OP(VSub);
|
||||
DEF_OP(VUQAdd);
|
||||
DEF_OP(VUQSub);
|
||||
DEF_OP(VSQAdd);
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
DEF_OP(VFMul);
|
||||
DEF_OP(VFDiv);
|
||||
DEF_OP(VFMin);
|
||||
DEF_OP(VFMax);
|
||||
DEF_OP(VFRecp);
|
||||
DEF_OP(VFSqrt);
|
||||
DEF_OP(VFRSqrt);
|
||||
DEF_OP(VNeg);
|
||||
DEF_OP(VFNeg);
|
||||
DEF_OP(VNot);
|
||||
DEF_OP(VUMin);
|
||||
DEF_OP(VSMin);
|
||||
DEF_OP(VUMax);
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
DEF_OP(VCMPGT);
|
||||
DEF_OP(VCMPGTZ);
|
||||
DEF_OP(VCMPLTZ);
|
||||
DEF_OP(VFCMPEQ);
|
||||
DEF_OP(VFCMPNEQ);
|
||||
DEF_OP(VFCMPLT);
|
||||
DEF_OP(VFCMPGT);
|
||||
DEF_OP(VFCMPLE);
|
||||
DEF_OP(VFCMPORD);
|
||||
DEF_OP(VFCMPUNO);
|
||||
DEF_OP(VUShl);
|
||||
DEF_OP(VUShr);
|
||||
DEF_OP(VSShr);
|
||||
DEF_OP(VUShlS);
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
DEF_OP(VUXTL2);
|
||||
DEF_OP(VSQXTN);
|
||||
DEF_OP(VSQXTN2);
|
||||
DEF_OP(VSQXTUN);
|
||||
DEF_OP(VSQXTUN2);
|
||||
DEF_OP(VUMul);
|
||||
DEF_OP(VUMull);
|
||||
DEF_OP(VSMul);
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
DEF_OP(AESEnc);
|
||||
DEF_OP(AESEncLast);
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
|
||||
///< F80 ops
|
||||
DEF_OP(F80LOADFCW);
|
||||
DEF_OP(F80ADD);
|
||||
DEF_OP(F80SUB);
|
||||
DEF_OP(F80MUL);
|
||||
DEF_OP(F80DIV);
|
||||
DEF_OP(F80FYL2X);
|
||||
DEF_OP(F80ATAN);
|
||||
DEF_OP(F80FPREM1);
|
||||
DEF_OP(F80FPREM);
|
||||
DEF_OP(F80SCALE);
|
||||
DEF_OP(F80CVT);
|
||||
DEF_OP(F80CVTINT);
|
||||
DEF_OP(F80CVTTO);
|
||||
DEF_OP(F80CVTTOINT);
|
||||
DEF_OP(F80ROUND);
|
||||
DEF_OP(F80F2XM1);
|
||||
DEF_OP(F80TAN);
|
||||
DEF_OP(F80SQRT);
|
||||
DEF_OP(F80SIN);
|
||||
DEF_OP(F80COS);
|
||||
DEF_OP(F80XTRACT_EXP);
|
||||
DEF_OP(F80XTRACT_SIG);
|
||||
DEF_OP(F80CMP);
|
||||
DEF_OP(F80BCDLOAD);
|
||||
DEF_OP(F80BCDSTORE);
|
||||
#undef DEF_OP
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
bool CompResult = false;
|
||||
switch (Cond) {
|
||||
case FEXCore::IR::COND_EQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_NEQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) != static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SGE:
|
||||
CompResult = static_cast<signed_type>(Src1) >= static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SLT:
|
||||
CompResult = static_cast<signed_type>(Src1) < static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SGT:
|
||||
CompResult = static_cast<signed_type>(Src1) > static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SLE:
|
||||
CompResult = static_cast<signed_type>(Src1) <= static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_UGE:
|
||||
CompResult = static_cast<unsigned_type>(Src1) >= static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_ULT:
|
||||
CompResult = static_cast<unsigned_type>(Src1) < static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_UGT:
|
||||
CompResult = static_cast<unsigned_type>(Src1) > static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_ULE:
|
||||
CompResult = static_cast<unsigned_type>(Src1) <= static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
|
||||
case FEXCore::IR::COND_FLU:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) < reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FGE:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) >= reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FLEU:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) <= reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FGT:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) > reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FU:
|
||||
CompResult = (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FNU:
|
||||
CompResult = !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
break;
|
||||
}
|
||||
|
||||
return CompResult;
|
||||
}
|
||||
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
|
||||
return IROp->Size;
|
||||
}
|
||||
|
||||
};
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,285 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static inline void CacheLineFlush(char *Addr) {
|
||||
#ifdef _M_X86_64
|
||||
__asm volatile (
|
||||
"clflush (%[Addr]);"
|
||||
:: [Addr] "r" (Addr)
|
||||
: "memory");
|
||||
#else
|
||||
__builtin___clear_cache(Addr, Addr+64);
|
||||
#endif
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (IROp->Size) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(FillRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
memset(GDP, 0, 16);
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint16_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint32_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint64_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
reinterpret_cast<std::atomic<uint8_t>*>(MemData)->store(*GetSrc<uint8_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
reinterpret_cast<std::atomic<uint16_t>*>(MemData)->store(*GetSrc<uint16_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
reinterpret_cast<std::atomic<uint32_t>*>(MemData)->store(*GetSrc<uint32_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
reinterpret_cast<std::atomic<uint64_t>*>(MemData)->store(*GetSrc<uint64_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Header.Args[0]);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[1]), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Header.Args[0]); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Header.Args[1])[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
STORE_DATA(1, uint8_t)
|
||||
STORE_DATA(2, uint16_t)
|
||||
STORE_DATA(4, uint32_t)
|
||||
STORE_DATA(8, uint64_t)
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
|
||||
}
|
||||
#undef STORE_DATA
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
|
||||
|
||||
// 64-byte cache line clear
|
||||
CacheLineFlush(MemData);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
uintptr_t MemData = *GetSrc<uintptr_t*>(Data->SSAData, Op->Addr);
|
||||
|
||||
// Force cacheline alignment
|
||||
MemData = MemData & ~(CPUIDEmu::CACHELINE_SIZE - 1);
|
||||
|
||||
using DataType = uint64_t;
|
||||
DataType *MemData64 = reinterpret_cast<DataType*>(MemData);
|
||||
|
||||
// 64-byte cache line zero
|
||||
for (size_t i = 0; i < (CPUIDEmu::CACHELINE_SIZE / sizeof(DataType)); ++i) {
|
||||
MemData64[i] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,153 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cstdint>
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val:
|
||||
std::atomic_thread_fence(std::memory_order_acquire);
|
||||
break;
|
||||
case IR::Fence_LoadStore.Val:
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
break;
|
||||
case IR::Fence_Store.Val:
|
||||
std::atomic_thread_fence(std::memory_order_release);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Halt: // HLT
|
||||
StopThread(Data->State);
|
||||
break;
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break Reason: {}", Op->Reason); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
uint32_t GuestRounding{};
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t Tmp{};
|
||||
__asm(R"(
|
||||
mrs %[Tmp], FPCR;
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp));
|
||||
// Extract the rounding
|
||||
// On ARM the ordering is different than on x86
|
||||
GuestRounding |= ((Tmp >> 24) & 1) ? IR::ROUND_MODE_FLUSH_TO_ZERO : 0;
|
||||
uint8_t RoundingMode = (Tmp >> 22) & 0b11;
|
||||
if (RoundingMode == 0)
|
||||
GuestRounding |= IR::ROUND_MODE_NEAREST;
|
||||
else if (RoundingMode == 1)
|
||||
GuestRounding |= IR::ROUND_MODE_POSITIVE_INFINITY;
|
||||
else if (RoundingMode == 2)
|
||||
GuestRounding |= IR::ROUND_MODE_NEGATIVE_INFINITY;
|
||||
else if (RoundingMode == 3)
|
||||
GuestRounding |= IR::ROUND_MODE_TOWARDS_ZERO;
|
||||
#else
|
||||
GuestRounding = _mm_getcsr();
|
||||
|
||||
// Extract the rounding
|
||||
GuestRounding = (GuestRounding >> 13) & 0b111;
|
||||
#endif
|
||||
memcpy(GDP, &GuestRounding, sizeof(GuestRounding));
|
||||
}
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
uint8_t GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t HostRounding{};
|
||||
__asm volatile(R"(
|
||||
mrs %[Tmp], FPCR;
|
||||
)"
|
||||
: [Tmp] "=r" (HostRounding));
|
||||
// Mask out the rounding
|
||||
HostRounding &= ~(0b111 << 22);
|
||||
|
||||
HostRounding |= (GuestRounding & IR::ROUND_MODE_FLUSH_TO_ZERO) ? (1U << 24) : 0;
|
||||
|
||||
uint8_t RoundingMode = GuestRounding & 0b11;
|
||||
if (RoundingMode == IR::ROUND_MODE_NEAREST)
|
||||
HostRounding |= (0b00U << 22);
|
||||
else if (RoundingMode == IR::ROUND_MODE_POSITIVE_INFINITY)
|
||||
HostRounding |= (0b01U << 22);
|
||||
else if (RoundingMode == IR::ROUND_MODE_NEGATIVE_INFINITY)
|
||||
HostRounding |= (0b10U << 22);
|
||||
else if (RoundingMode == IR::ROUND_MODE_TOWARDS_ZERO)
|
||||
HostRounding |= (0b11U << 22);
|
||||
|
||||
__asm volatile(R"(
|
||||
msr FPCR, %[Tmp];
|
||||
)"
|
||||
:: [Tmp] "r" (HostRounding));
|
||||
#else
|
||||
uint32_t HostRounding = _mm_getcsr();
|
||||
|
||||
// Cut out the host rounding mode
|
||||
HostRounding &= ~(0b111 << 13);
|
||||
|
||||
// Insert our new rounding mode
|
||||
HostRounding |= GuestRounding << 13;
|
||||
_mm_setcsr(HostRounding);
|
||||
#endif
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (OpSize <= 8) {
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
|
||||
}
|
||||
else if (OpSize == 16) {
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src0 = Src;
|
||||
uint64_t Src1 = Src >> 64;
|
||||
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
|
||||
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
|
||||
}
|
||||
else
|
||||
LOGMAN_MSG_A_FMT("Unknown value size: {}", OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
uint32_t CPU, CPUNode;
|
||||
FHU::Syscalls::getcpu(&CPU, &CPUNode);
|
||||
GD = (CPUNode << 12) | CPU;
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,42 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
uintptr_t Src = GetSrc<uintptr_t>(Data->SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP,
|
||||
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
|
||||
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
|
||||
|
||||
memcpy(Dst, Src_Lower, Op->Header.Size);
|
||||
memcpy(Dst + Op->Header.Size, Src_Upper, Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
+149
-6
@@ -8,6 +8,10 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
@@ -34,7 +38,7 @@ static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
@@ -61,7 +65,13 @@ DEF_OP(EntrypointOffset) {
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
LoadConstant(Dst, Constant);
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
LoadConstant(Dst, Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -80,8 +90,6 @@ DEF_OP(CycleCounter) {
|
||||
#endif
|
||||
}
|
||||
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
DEF_OP(Add) {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -163,7 +171,7 @@ DEF_OP(Mul) {
|
||||
case 8:
|
||||
mul(Dst, GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Mul size: %d", OpSize);
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Mul size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -390,6 +398,19 @@ DEF_OP(And) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const auto& Lhs = Op->Header.Args[0];
|
||||
const auto& Rhs = Op->Header.Args[1];
|
||||
uint64_t Const{};
|
||||
|
||||
if (IsInlineConstant(Rhs, &Const)) {
|
||||
bic(GRS(Node), GRS(Lhs.ID()), Const);
|
||||
} else {
|
||||
bic(GRS(Node), GRS(Lhs.ID()), GRS(Rhs.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Xor) {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint64_t Const;
|
||||
@@ -498,6 +519,125 @@ DEF_OP(Extr) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Register Input = GRS(Op->Args(0).ID());
|
||||
const Register Mask = GRS(Op->Args(1).ID());
|
||||
const Register Dest = GRS(Node);
|
||||
|
||||
const Register ShiftedBitReg = OpSize <= 4 ? TMP1.W() : TMP1;
|
||||
const Register BitReg = OpSize <= 4 ? TMP2.W() : TMP2;
|
||||
const Register SubMaskReg = OpSize <= 4 ? TMP3.W() : TMP3;
|
||||
const Register IndexReg = OpSize <= 4 ? TMP4.W() : TMP4;
|
||||
const Register SizedZero = OpSize <= 4 ? Register{wzr} : Register{xzr};
|
||||
|
||||
const Register InputReg = OpSize <= 4 ? SRA64[0].W() : SRA64[0];
|
||||
const Register MaskReg = OpSize <= 4 ? SRA64[1].W() : SRA64[1];
|
||||
const Register DestReg = OpSize <= 4 ? SRA64[2].W() : SRA64[2];
|
||||
const auto SpillCode = 1U << InputReg.GetCode() |
|
||||
1U << MaskReg.GetCode() |
|
||||
1U << DestReg.GetCode();
|
||||
|
||||
aarch64::Label EarlyExit;
|
||||
aarch64::Label NextBit;
|
||||
aarch64::Label Done;
|
||||
|
||||
cbz(Mask, &EarlyExit);
|
||||
mov(IndexReg, SizedZero);
|
||||
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, SpillCode);
|
||||
mov(InputReg, Input);
|
||||
mov(MaskReg, Mask);
|
||||
mov(DestReg, SizedZero);
|
||||
|
||||
// Main loop
|
||||
bind(&NextBit);
|
||||
rbit(ShiftedBitReg, MaskReg);
|
||||
clz(ShiftedBitReg, ShiftedBitReg);
|
||||
lsrv(BitReg, InputReg, IndexReg);
|
||||
and_(BitReg, BitReg, 1);
|
||||
sub(SubMaskReg, MaskReg, 1);
|
||||
add(IndexReg, IndexReg, 1);
|
||||
ands(MaskReg, MaskReg, SubMaskReg);
|
||||
lslv(ShiftedBitReg, BitReg, ShiftedBitReg);
|
||||
orr(DestReg, DestReg, ShiftedBitReg);
|
||||
b(&NextBit, Condition::ne);
|
||||
// Store result in a temp so it doesn't get clobbered.
|
||||
// and restore it after the re-fill below.
|
||||
mov(IndexReg, DestReg);
|
||||
// Restore our registers before leaving
|
||||
// TODO: Also remove along with above TODO.
|
||||
FillStaticRegs(false, SpillCode);
|
||||
mov(Dest, IndexReg);
|
||||
b(&Done);
|
||||
|
||||
// Early exit
|
||||
bind(&EarlyExit);
|
||||
mov(Dest, SizedZero);
|
||||
|
||||
// All done with nothing to do.
|
||||
bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(PExt) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Register Input = GRS(Op->Args(0).ID());
|
||||
const Register Mask = GRS(Op->Args(1).ID());
|
||||
const Register Dest = GRS(Node);
|
||||
|
||||
const Register MaskReg = OpSize <= 4 ? TMP1.W() : TMP1;
|
||||
const Register BitReg = OpSize <= 4 ? TMP2.W() : TMP2;
|
||||
const Register SubMaskReg = OpSize <= 4 ? TMP3.W() : TMP3;
|
||||
const Register Offset = OpSize <= 4 ? TMP4.W() : TMP4;
|
||||
const Register SizedZero = OpSize <= 4 ? Register{wzr} : Register{xzr};
|
||||
|
||||
aarch64::Label EarlyExit;
|
||||
aarch64::Label NextBit;
|
||||
aarch64::Label Done;
|
||||
|
||||
cbz(Mask, &EarlyExit);
|
||||
mov(MaskReg, Mask);
|
||||
mov(Offset, SizedZero);
|
||||
|
||||
// We sadly need to spill a reg for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, 1U << Mask.GetCode());
|
||||
mov(Mask, SizedZero);
|
||||
|
||||
// Main loop
|
||||
bind(&NextBit);
|
||||
rbit(BitReg, MaskReg);
|
||||
clz(BitReg, BitReg);
|
||||
sub(SubMaskReg, MaskReg, 1);
|
||||
ands(MaskReg, SubMaskReg, MaskReg);
|
||||
lsrv(BitReg, Input, BitReg);
|
||||
and_(BitReg, BitReg, 1);
|
||||
lslv(BitReg, BitReg, Offset);
|
||||
add(Offset, Offset, 1);
|
||||
orr(Mask, BitReg, Mask);
|
||||
b(&NextBit, Condition::ne);
|
||||
mov(Dest, Mask);
|
||||
// Restore our mask register before leaving
|
||||
// TODO: Also remove along with above TODO.
|
||||
FillStaticRegs(false, 1U << Mask.GetCode());
|
||||
b(&Done);
|
||||
|
||||
// Early exit
|
||||
bind(&EarlyExit);
|
||||
mov(Dest, SizedZero);
|
||||
|
||||
// All done with nothing to do.
|
||||
bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -903,7 +1043,7 @@ Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FLU: return Condition::lt;
|
||||
case FEXCore::IR::COND_FGE: return Condition::ge;
|
||||
case FEXCore::IR::COND_FLEU:return Condition::le;
|
||||
case FEXCore::IR::COND_FGT: return Condition::hi;
|
||||
case FEXCore::IR::COND_FGT: return Condition::gt;
|
||||
case FEXCore::IR::COND_FU: return Condition::vs;
|
||||
case FEXCore::IR::COND_FNU: return Condition::vc;
|
||||
case FEXCore::IR::COND_VS:
|
||||
@@ -1079,12 +1219,15 @@ void Arm64JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
|
||||
+62
-68
@@ -9,7 +9,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -19,7 +19,7 @@ DEF_OP(CASPair) {
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP3, Expected.first);
|
||||
mov(TMP4, Expected.second);
|
||||
|
||||
@@ -44,15 +44,12 @@ DEF_OP(CASPair) {
|
||||
aarch64::Label LoopNotExpected;
|
||||
aarch64::Label LoopExpected;
|
||||
bind(&LoopTop);
|
||||
nop();
|
||||
|
||||
ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
nop();
|
||||
cmp(TMP2.W(), Expected.first.W());
|
||||
ccmp(TMP3.W(), Expected.second.W(), NoFlag, Condition::eq);
|
||||
b(&LoopNotExpected, Condition::ne);
|
||||
nop();
|
||||
stlxp(TMP2.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc));
|
||||
nop();
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
mov(Dst.first.W(), Expected.first.W());
|
||||
mov(Dst.second.W(), Expected.second.W());
|
||||
@@ -73,15 +70,12 @@ DEF_OP(CASPair) {
|
||||
aarch64::Label LoopNotExpected;
|
||||
aarch64::Label LoopExpected;
|
||||
bind(&LoopTop);
|
||||
nop();
|
||||
|
||||
ldaxp(TMP2.X(), TMP3.X(), MemOperand(MemSrc));
|
||||
nop();
|
||||
cmp(TMP2.X(), Expected.first.X());
|
||||
ccmp(TMP3.X(), Expected.second.X(), NoFlag, Condition::eq);
|
||||
b(&LoopNotExpected, Condition::ne);
|
||||
nop();
|
||||
stlxp(TMP2.X(), Desired.first.X(), Desired.second.X(), MemOperand(MemSrc));
|
||||
nop();
|
||||
cbnz(TMP2.X(), &LoopTop);
|
||||
mov(Dst.first.X(), Expected.first.X());
|
||||
mov(Dst.second.X(), Expected.second.X());
|
||||
@@ -116,7 +110,7 @@ DEF_OP(CAS) {
|
||||
auto Desired = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, Expected);
|
||||
switch (OpSize) {
|
||||
case 1: casalb(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
|
||||
@@ -224,18 +218,18 @@ DEF_OP(AtomicAdd) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -272,7 +266,7 @@ DEF_OP(AtomicAdd) {
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -282,19 +276,19 @@ DEF_OP(AtomicSub) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(TMP2.X(), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -331,7 +325,7 @@ DEF_OP(AtomicSub) {
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -341,19 +335,19 @@ DEF_OP(AtomicAnd) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 2: stclrlh(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 4: stclrl(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 8: stclrl(TMP2.X(), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -390,7 +384,7 @@ DEF_OP(AtomicAnd) {
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -400,18 +394,18 @@ DEF_OP(AtomicOr) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: stsetl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: stsetl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -448,7 +442,7 @@ DEF_OP(AtomicOr) {
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -458,18 +452,18 @@ DEF_OP(AtomicXor) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: steorl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: steorl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -506,7 +500,7 @@ DEF_OP(AtomicXor) {
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -516,19 +510,19 @@ DEF_OP(AtomicSwap) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -565,7 +559,7 @@ DEF_OP(AtomicSwap) {
|
||||
mov(GetReg<RA_64>(Node), TMP2.X());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -574,18 +568,18 @@ DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -626,7 +620,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -635,19 +629,19 @@ DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -688,7 +682,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -697,19 +691,19 @@ DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldclralh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldclral(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldclral(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -750,7 +744,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -759,18 +753,18 @@ DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldsetal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldsetal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -811,7 +805,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -820,18 +814,18 @@ DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldeoral(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldeoral(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -872,7 +866,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -882,7 +876,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
// TMP2-TMP3
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
@@ -923,7 +917,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+160
-37
@@ -11,12 +11,13 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
@@ -105,23 +106,16 @@ DEF_OP(ExitFunction) {
|
||||
}
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
|
||||
Label *TargetLabel;
|
||||
auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID());
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second;
|
||||
}
|
||||
else {
|
||||
TargetLabel = &IsTarget->second;
|
||||
}
|
||||
PendingTargetLabel = TargetLabel;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
#define GRFCMP(Node) (Op->CompareSize == 4 ? GetDst(Node).S() : GetDst(Node).D())
|
||||
|
||||
Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
static Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return Condition::eq;
|
||||
case FEXCore::IR::COND_NEQ: return Condition::ne;
|
||||
@@ -136,7 +130,7 @@ Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FLU: return Condition::lt;
|
||||
case FEXCore::IR::COND_FGE: return Condition::ge;
|
||||
case FEXCore::IR::COND_FLEU:return Condition::le;
|
||||
case FEXCore::IR::COND_FGT: return Condition::hi;
|
||||
case FEXCore::IR::COND_FGT: return Condition::gt;
|
||||
case FEXCore::IR::COND_FU: return Condition::vs;
|
||||
case FEXCore::IR::COND_FNU: return Condition::vc;
|
||||
case FEXCore::IR::COND_VS:
|
||||
@@ -153,22 +147,10 @@ Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
Label *TrueTargetLabel;
|
||||
Label *FalseTargetLabel;
|
||||
|
||||
auto TrueIter = JumpTargets.find(Op->TrueBlock.ID());
|
||||
auto FalseIter = JumpTargets.find(Op->FalseBlock.ID());
|
||||
|
||||
if (TrueIter == JumpTargets.end()) {
|
||||
TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
}
|
||||
else {
|
||||
TrueTargetLabel = &TrueIter->second;
|
||||
}
|
||||
|
||||
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
uint64_t Const;
|
||||
bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
@@ -178,10 +160,11 @@ DEF_OP(CondJump) {
|
||||
cbnz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (isConst)
|
||||
if (isConst) {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), Const);
|
||||
else
|
||||
} else {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(GRFCMP(Op->Cmp1.ID()), GRFCMP(Op->Cmp2.ID()));
|
||||
} else {
|
||||
@@ -191,13 +174,7 @@ DEF_OP(CondJump) {
|
||||
b(TrueTargetLabel, MapBranchCC(Op->Cond));
|
||||
}
|
||||
|
||||
if (FalseIter == JumpTargets.end()) {
|
||||
FalseTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
}
|
||||
else {
|
||||
FalseTargetLabel = &FalseIter->second;
|
||||
}
|
||||
PendingTargetLabel = FalseTargetLabel;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
@@ -235,6 +212,152 @@ DEF_OP(Syscall) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X8: SyscallNumber - RA INTERSECT
|
||||
// X0: Arg0 & Return
|
||||
// X1: Arg1
|
||||
// X2: Arg2
|
||||
// X3: Arg3
|
||||
// X4: Arg4 - RA INTERSECT
|
||||
// X5: Arg5 - RA INTERSECT
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<vixl::aarch64::Register, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
|
||||
x0, x1, x2, x3, x4, x5
|
||||
}};
|
||||
|
||||
bool Intersects{};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
if (Reg.GetCode() == x8.GetCode() ||
|
||||
Reg.GetCode() == x4.GetCode() ||
|
||||
Reg.GetCode() == x5.GetCode()) {
|
||||
|
||||
SpillMask |= (1U << Reg.GetCode());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
// XXX: For some reason spilling only the x4, x5, and x8 registers was causing issues
|
||||
// Come back to this once investigation reveals why it fails the gvisor ioctl test
|
||||
// For now override to all GPRs
|
||||
SpillMask = ~0U;
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.GetCode() == x8.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == x4.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == x5.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.GetCode() == x8.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == x4.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == x5.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
mov(RegArgs[i], Reg);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Reg = GetReg<RA_32>(Op->Header.Args[i].ID());
|
||||
if (Reg.GetCode() == w8.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == w4.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == w5.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i].W(), Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(x8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
else {
|
||||
uxtw(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
// Arguments are passed as follows:
|
||||
@@ -256,7 +379,6 @@ DEF_OP(Thunk) {
|
||||
FillStaticRegs(); // load from ctx after ra64 refill
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
|
||||
@@ -370,6 +492,7 @@ void Arm64JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
@@ -39,11 +39,11 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
|
||||
@@ -10,7 +10,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
@@ -54,7 +54,7 @@ DEF_OP(AESDecLast) {
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Label Constant;
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label PastConstant;
|
||||
|
||||
// Do a "regular" AESE step
|
||||
@@ -63,8 +63,7 @@ DEF_OP(AESKeyGenAssist) {
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
adr(TMP1.X(), &Constant);
|
||||
ldr(VTMP3, MemOperand(TMP1.X()));
|
||||
ldr(VTMP3, &ConstantLiteral);
|
||||
|
||||
// Now EOR in the RCON
|
||||
if (Op->RCON) {
|
||||
@@ -80,14 +79,29 @@ DEF_OP(AESKeyGenAssist) {
|
||||
}
|
||||
|
||||
b(&PastConstant);
|
||||
bind(&Constant);
|
||||
dc32(0x0B0E0104);
|
||||
dc32(0x040B0E01);
|
||||
dc32(0x0306090C);
|
||||
dc32(0x0C030609);
|
||||
place(&ConstantLiteral);
|
||||
bind(&PastConstant);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32cb(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 2:
|
||||
crc32ch(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 4:
|
||||
crc32cw(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_64>(Op->Src2.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -97,7 +111,7 @@ void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
|
||||
|
||||
+59
-192
@@ -42,7 +42,7 @@ void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -55,7 +55,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x1);
|
||||
@@ -112,7 +112,12 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
else {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x1);
|
||||
@@ -133,7 +138,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -153,7 +158,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -173,7 +178,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -192,7 +197,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -211,7 +216,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -230,10 +235,10 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -252,7 +257,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -273,10 +278,10 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
@@ -302,7 +307,7 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
|
||||
@@ -316,6 +321,9 @@ Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
@@ -324,165 +332,15 @@ void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
if (!Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
// 1 = 16bit
|
||||
// 2 = 32bit
|
||||
// 3 = 64bit
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1011'0000'0000; // Inner shareable all
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if (ParanoidTSO()) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t LDR = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
if (ParanoidTSO()) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXP_MASK) == FEXCore::ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
uint32_t DataReg2 = (Instr >> 10) & 0x1F;
|
||||
// Convert to LDP
|
||||
uint32_t LDP = 0b0010'1001'0100'0000'0000'0000'0000'0000;
|
||||
LDP |= Size << 31;
|
||||
LDP |= DataReg2 << 10;
|
||||
LDP |= AddrReg << 5;
|
||||
LDP |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDP;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::STLXP_MASK) == FEXCore::ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
uint32_t DataReg2 = (Instr >> 10) & 0x1F;
|
||||
// Convert to STP
|
||||
uint32_t STP = 0b0010'1001'0000'0000'0000'0000'0000'0000;
|
||||
STP |= Size << 31;
|
||||
STP |= DataReg2 << 10;
|
||||
STP |= AddrReg << 5;
|
||||
STP |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STP;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: {} Instruction: 0x{:08x}\n", Op, fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXR_MASK) == FEXCore::ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ucontext, info);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + BytesToSkip);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXR: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[-1], 16);
|
||||
return true;
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: Arm64Emitter(0)
|
||||
: Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
{
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = true;
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
@@ -496,7 +354,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
|
||||
RAPass = Thread->PassManager->GetRAPass();
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
#if DEBUG
|
||||
Decoder.AppendVisitor(&Disasm)
|
||||
@@ -538,6 +396,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
if (!CompileThread) {
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalReturnInstruction = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
@@ -548,7 +408,13 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->HandleSIGBUS(Signal, info, ucontext);
|
||||
|
||||
if (!Core->Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Core->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
@@ -561,7 +427,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
@@ -618,7 +484,7 @@ Arm64JITCore::~Arm64JITCore() {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(uint32_t Node) const {
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
@@ -627,7 +493,7 @@ IR::PhysicalRegister Arm64JITCore::GetPhys(uint32_t Node) const {
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) const {
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
@@ -642,7 +508,7 @@ aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) const
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) const {
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
@@ -657,18 +523,18 @@ aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) const
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(uint32_t Node) const {
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
return RA32Pair[Reg];
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(uint32_t Node) const {
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
return RA64Pair[Reg];
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) const {
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
@@ -682,7 +548,7 @@ aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) const {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetDst(uint32_t Node) const {
|
||||
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
@@ -716,7 +582,12 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = Entry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
@@ -724,18 +595,18 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(uint32_t Node) const {
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
|
||||
}
|
||||
|
||||
|
||||
bool Arm64JITCore::IsFPR(uint32_t Node) const {
|
||||
bool Arm64JITCore::IsFPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPR(uint32_t Node) const {
|
||||
bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
@@ -781,8 +652,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
auto Buffer = GetBuffer();
|
||||
auto GuestEntry = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
auto GuestEntry = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
aarch64::Label RunBlock;
|
||||
@@ -809,7 +679,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
bind(&RunBlock);
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -832,11 +702,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
#endif
|
||||
|
||||
{
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
@@ -849,11 +716,11 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()), 0, IR->GetID(BlockNode)});
|
||||
DebugData->Subblocks.push_back({GetCursorAddress<uintptr_t>(), 0, IR->GetID(BlockNode)});
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
@@ -861,7 +728,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.back().HostCodeSize = Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()) - DebugData->Subblocks.back().HostCodeStart;
|
||||
DebugData->Subblocks.back().HostCodeSize = GetCursorAddress<uintptr_t>() - DebugData->Subblocks.back().HostCodeStart;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -874,7 +741,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
auto CodeEnd = GetCursorAddress<uint64_t>();
|
||||
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(GuestEntry), CodeEnd - reinterpret_cast<uint64_t>(GuestEntry));
|
||||
|
||||
if (DebugData) {
|
||||
|
||||
+49
-29
@@ -42,24 +42,31 @@ public:
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
~Arm64JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(uint64_t Entry, FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
|
||||
bool NeedsOpDispatch() override { return true; }
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
@@ -71,7 +78,7 @@ private:
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
|
||||
std::map<IR::OrderedNodeWrapper::NodeOffsetType, aarch64::Label> JumpTargets;
|
||||
std::map<IR::NodeID, aarch64::Label> JumpTargets;
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
@@ -95,35 +102,39 @@ private:
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
template<uint8_t RAType>
|
||||
aarch64::Register GetReg(uint32_t Node) const;
|
||||
[[nodiscard]] aarch64::Register GetReg(IR::NodeID Node) const;
|
||||
|
||||
template<>
|
||||
aarch64::Register GetReg<RA_32>(uint32_t Node) const;
|
||||
[[nodiscard]] aarch64::Register GetReg<RA_32>(IR::NodeID Node) const;
|
||||
template<>
|
||||
aarch64::Register GetReg<RA_64>(uint32_t Node) const;
|
||||
[[nodiscard]] aarch64::Register GetReg<RA_64>(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair(uint32_t Node) const;
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair(IR::NodeID Node) const;
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(uint32_t Node) const;
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(IR::NodeID Node) const;
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(uint32_t Node) const;
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(IR::NodeID Node) const;
|
||||
|
||||
aarch64::VRegister GetSrc(uint32_t Node) const;
|
||||
aarch64::VRegister GetDst(uint32_t Node) const;
|
||||
[[nodiscard]] aarch64::VRegister GetSrc(IR::NodeID Node) const;
|
||||
[[nodiscard]] aarch64::VRegister GetDst(IR::NodeID Node) const;
|
||||
|
||||
FEXCore::IR::RegisterClassType GetRegClass(uint32_t Node) const;
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node) const;
|
||||
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const;
|
||||
|
||||
bool IsFPR(uint32_t Node) const;
|
||||
bool IsGPR(uint32_t Node) const;
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
|
||||
MemOperand GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
[[nodiscard]] MemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
aarch64::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
struct LiveRange {
|
||||
uint32_t Begin;
|
||||
@@ -164,6 +175,8 @@ private:
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalReturnInstruction{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
@@ -173,8 +186,8 @@ private:
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(FEXCore::IR::IROp_Header *IROp, uint32_t Node);
|
||||
std::array<OpHandler, FEXCore::IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
void RegisterBranchHandlers();
|
||||
@@ -185,7 +198,7 @@ private:
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
#define DEF_OP(x) void Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
@@ -213,6 +226,7 @@ private:
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
@@ -220,6 +234,8 @@ private:
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
@@ -268,6 +284,7 @@ private:
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
@@ -288,7 +305,7 @@ private:
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
@@ -307,6 +324,7 @@ private:
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
@@ -317,6 +335,7 @@ private:
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -423,6 +442,7 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
+68
-47
@@ -11,7 +11,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
@@ -261,7 +261,7 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = Op->Size;
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -288,7 +288,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
ldr(GetReg<RA_64>(Node), MemOperand(TMP1, Op->BaseOffset));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -335,7 +335,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -349,7 +349,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
size_t size = Op->Size;
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -378,7 +378,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
str(value, MemOperand(TMP1, Op->BaseOffset));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -427,7 +427,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -571,11 +571,11 @@ DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GenerateMemOperand(Op->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
ldrb(Dst, MemSrc);
|
||||
break;
|
||||
@@ -588,12 +588,12 @@ DEF_OP(LoadMem) {
|
||||
case 8:
|
||||
ldr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Dst = GetDst(Node);
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
ldr(Dst.B(), MemSrc);
|
||||
break;
|
||||
@@ -609,7 +609,7 @@ DEF_OP(LoadMem) {
|
||||
case 16:
|
||||
ldr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -623,8 +623,8 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("LoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldaprb(Dst, MemSrc);
|
||||
@@ -633,7 +633,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Aligned
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
nop();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldaprh(Dst, MemSrc);
|
||||
break;
|
||||
@@ -643,13 +643,13 @@ DEF_OP(LoadMemTSO) {
|
||||
case 8:
|
||||
ldapr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldarb(Dst, MemSrc);
|
||||
@@ -658,7 +658,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Aligned
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
nop();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldarh(Dst, MemSrc);
|
||||
break;
|
||||
@@ -668,7 +668,7 @@ DEF_OP(LoadMemTSO) {
|
||||
case 8:
|
||||
ldar(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
@@ -676,7 +676,7 @@ DEF_OP(LoadMemTSO) {
|
||||
else {
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
auto Dst = GetDst(Node);
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldr(Dst.H(), MemSrc);
|
||||
break;
|
||||
@@ -689,7 +689,7 @@ DEF_OP(LoadMemTSO) {
|
||||
case 16:
|
||||
ldr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
}
|
||||
@@ -700,10 +700,10 @@ DEF_OP(StoreMem) {
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
auto MemSrc = GenerateMemOperand(Op->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
strb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
@@ -716,12 +716,12 @@ DEF_OP(StoreMem) {
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
str(Src.B(), MemSrc);
|
||||
break;
|
||||
@@ -737,7 +737,7 @@ DEF_OP(StoreMem) {
|
||||
case 16:
|
||||
str(Src, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -751,13 +751,13 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
@@ -767,7 +767,7 @@ DEF_OP(StoreMemTSO) {
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
@@ -775,7 +775,7 @@ DEF_OP(StoreMemTSO) {
|
||||
else {
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
str(Src.B(), MemSrc);
|
||||
break;
|
||||
@@ -791,7 +791,7 @@ DEF_OP(StoreMemTSO) {
|
||||
case 16:
|
||||
str(Src, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
}
|
||||
@@ -807,14 +807,14 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldarb(Dst, MemSrc);
|
||||
}
|
||||
else {
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldarh(Dst, MemSrc);
|
||||
break;
|
||||
@@ -824,13 +824,13 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
case 8:
|
||||
ldar(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Dst = GetDst(Node);
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldarh(TMP1.W(), MemSrc);
|
||||
fmov(Dst.H(), TMP1.W());
|
||||
@@ -850,7 +850,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
mov(Dst.V2D(), 0, TMP1);
|
||||
mov(Dst.V2D(), 1, TMP2);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -864,12 +864,12 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
@@ -879,19 +879,19 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
if (Op->Size == 1) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
mov(TMP1.W(), Src.V16B(), 0);
|
||||
stlrb(TMP1, MemSrc);
|
||||
}
|
||||
else {
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
mov(TMP1.W(), Src.V8H(), 0);
|
||||
stlrh(TMP1, MemSrc);
|
||||
@@ -911,15 +911,13 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
Label B;
|
||||
bind(&B);
|
||||
|
||||
nop(); // < Overwritten with DMB
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS
|
||||
nop(); // < Overwritten with DMB
|
||||
ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
|
||||
cbnz(TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -941,13 +939,35 @@ DEF_OP(CacheLineClear) {
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
mov(TMP1, MemReg);
|
||||
for (size_t i = 0; i < std::max(1U, DCacheLineSize / 64U); ++i) {
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
dc(DataCacheOp::CVAU, TMP1);
|
||||
add(TMP1, TMP1, DCacheLineSize);
|
||||
add(TMP1, TMP1, CTX->HostFeatures.DCacheLineSize);
|
||||
}
|
||||
dsb(InnerShareable, BarrierAll);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
dc(DataCacheOp::ZVA, MemReg);
|
||||
}
|
||||
else {
|
||||
// We must walk the cacheline ourselves
|
||||
// Force cacheline alignment
|
||||
and_(TMP1, MemReg, ~(CPUIDEmu::CACHELINE_SIZE - 1));
|
||||
// This will end up being four STPs
|
||||
// Depending on uarch it could be slightly more efficient in instructions emitted
|
||||
// and uops to use vector pair STP, but we want the non-temporal bit specifically here
|
||||
for (size_t i = 0; i < CPUIDEmu::CACHELINE_SIZE; i += 16) {
|
||||
stnp(xzr, xzr, MemOperand(TMP1, i, Offset));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -974,6 +994,7 @@ void Arm64JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+71
-13
@@ -17,7 +17,7 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
@@ -38,20 +38,16 @@ DEF_OP(Fence) {
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case 0: // Hard fault
|
||||
case 5: // Guest ud2
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
hlt(4);
|
||||
break;
|
||||
case 1: // Int <imm8>
|
||||
hlt(4);
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
ResetStack();
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
|
||||
br(TMP1);
|
||||
break;
|
||||
case 2: // overflow
|
||||
hlt(4);
|
||||
break;
|
||||
case 3: // int 1
|
||||
hlt(4);
|
||||
break;
|
||||
case 4: { // HLT
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
@@ -62,13 +58,22 @@ DEF_OP(Break) {
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case 6: { // INT3
|
||||
case FEXCore::IR::Break_Interrupt3: { // INT3
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
|
||||
br(TMP1);
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
}
|
||||
}
|
||||
@@ -153,6 +158,58 @@ DEF_OP(Print) {
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Allocate some temporary space for storing the uint32_t CPU and Node IDs
|
||||
sub(sp, sp, 16);
|
||||
|
||||
// Load the getcpu syscall number
|
||||
LoadConstant(x8, SYS_getcpu);
|
||||
|
||||
// CPU pointer in x0
|
||||
add(x0, sp, 0);
|
||||
// Node in x1
|
||||
add(x1, sp, 4);
|
||||
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
// Load the values returned by the kernel
|
||||
ldp(w0, w1, MemOperand(sp));
|
||||
// Deallocate stack space
|
||||
sub(sp, sp, 16);
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
|
||||
// Now store the result in the destination in the expected format
|
||||
// uint32_t Res = (node << 12) | cpu;
|
||||
// CPU is in w0
|
||||
// Node is in w1
|
||||
orr(GetReg<RA_64>(Node), x0, Operand(x1, LSL, 12));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -169,6 +226,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VectorZero) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch (OpSize) {
|
||||
|
||||
+8
-3
@@ -13,6 +13,11 @@ struct InternalThreadState;
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
}
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -5,10 +5,22 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
@@ -33,7 +45,13 @@ DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
mov(GetDst<RA_64>(Node), Constant);
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
mov(GetDst<RA_64>(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -410,6 +428,25 @@ DEF_OP(And) {
|
||||
mov(Dst, rax);
|
||||
}
|
||||
|
||||
DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const auto& Lhs = Op->Header.Args[0];
|
||||
const auto& Rhs = Op->Header.Args[1];
|
||||
auto Dst = GRD(Node);
|
||||
|
||||
uint64_t Const{};
|
||||
if (IsInlineConstant(Rhs, &Const)) {
|
||||
mov(Dst, GRS(Lhs.ID()));
|
||||
and_(Dst, ~Const);
|
||||
} else {
|
||||
const auto Temp = IROp->Size <= 4 ? Xbyak::Reg{rax.cvt32()} : Xbyak::Reg{rax};
|
||||
mov(Temp, GRS(Rhs.ID()));
|
||||
not_(Temp);
|
||||
and_(Temp, GRS(Lhs.ID()));
|
||||
mov(Dst, Temp);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Xor) {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
@@ -640,6 +677,36 @@ DEF_OP(Extr) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Input = GRS(Op->Args(0).ID());
|
||||
const auto Mask = GRS(Op->Args(1).ID());
|
||||
const auto Dest = GRD(Node);
|
||||
|
||||
if (OpSize == 4) {
|
||||
pdep(Dest.cvt32(), Input.cvt32(), Mask.cvt32());
|
||||
} else {
|
||||
pdep(Dest.cvt64(), Input.cvt64(), Mask.cvt64());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PExt) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Input = GRS(Op->Args(0).ID());
|
||||
const auto Mask = GRS(Op->Args(1).ID());
|
||||
const auto Dest = GRD(Node);
|
||||
|
||||
if (OpSize == 4) {
|
||||
pext(Dest.cvt32(), Input.cvt32(), Mask.cvt32());
|
||||
} else {
|
||||
pext(Dest.cvt64(), Input.cvt64(), Mask.cvt64());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -1041,10 +1108,6 @@ DEF_OP(Sbfe) {
|
||||
}
|
||||
}
|
||||
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
auto Dst = GRD(Node);
|
||||
@@ -1214,12 +1277,15 @@ void X86JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
|
||||
+33
-26
@@ -5,10 +5,17 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -114,7 +121,7 @@ DEF_OP(AtomicAdd) {
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
add(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
@@ -127,7 +134,7 @@ DEF_OP(AtomicAdd) {
|
||||
case 8:
|
||||
add(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -136,7 +143,7 @@ DEF_OP(AtomicSub) {
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
sub(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
@@ -149,7 +156,7 @@ DEF_OP(AtomicSub) {
|
||||
case 8:
|
||||
sub(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -158,7 +165,7 @@ DEF_OP(AtomicAnd) {
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
and_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
@@ -171,7 +178,7 @@ DEF_OP(AtomicAnd) {
|
||||
case 8:
|
||||
and_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,7 +187,7 @@ DEF_OP(AtomicOr) {
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
or_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
@@ -193,7 +200,7 @@ DEF_OP(AtomicOr) {
|
||||
case 8:
|
||||
or_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -202,7 +209,7 @@ DEF_OP(AtomicXor) {
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
xor_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
@@ -215,7 +222,7 @@ DEF_OP(AtomicXor) {
|
||||
case 8:
|
||||
xor_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,7 +232,7 @@ DEF_OP(AtomicSwap) {
|
||||
Xbyak::Reg MemReg = rax;
|
||||
mov(MemReg, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
movzx(GetDst<RA_64>(Node), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
lock();
|
||||
@@ -246,7 +253,7 @@ DEF_OP(AtomicSwap) {
|
||||
lock();
|
||||
xchg(qword [MemReg], GetDst<RA_64>(Node));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicSwap size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicSwap size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -254,7 +261,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
movzx(rcx, GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
lock();
|
||||
@@ -279,7 +286,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
xadd(qword [MemReg], rcx);
|
||||
mov(GetDst<RA_64>(Node), rcx);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAdd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAdd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -287,7 +294,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
mov(cl, GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
neg(cl);
|
||||
@@ -316,7 +323,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
xadd(qword [MemReg], rcx);
|
||||
mov(GetDst<RA_64>(Node), rcx);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchSub size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchSub size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -326,7 +333,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
// TMP1 = rax
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
|
||||
@@ -394,7 +401,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
mov(GetDst<RA_64>(Node), TMP3.cvt64());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAnd size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAnd size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -403,7 +410,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
|
||||
// TMP1 = rax
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
|
||||
@@ -471,7 +478,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
mov(GetDst<RA_64>(Node), TMP3.cvt64());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchOr size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchOr size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -480,7 +487,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
|
||||
// TMP1 = rax
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
|
||||
@@ -548,7 +555,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
mov(GetDst<RA_64>(Node), TMP3.cvt64());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchXor size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchXor size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -556,7 +563,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
mov(TMP1.cvt8(), byte [MemReg]);
|
||||
|
||||
@@ -624,7 +631,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
mov(GetDst<RA_64>(Node), TMP3.cvt64());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchNeg size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchNeg size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+26
-37
@@ -4,17 +4,31 @@ tags: backend|x86-64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
@@ -114,18 +128,10 @@ DEF_OP(ExitFunction) {
|
||||
}
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
|
||||
Label *TargetLabel;
|
||||
auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID());
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second;
|
||||
}
|
||||
else {
|
||||
TargetLabel = &IsTarget->second;
|
||||
}
|
||||
|
||||
PendingTargetLabel = TargetLabel;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
@@ -133,18 +139,7 @@ DEF_OP(Jump) {
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
Label *TrueTargetLabel;
|
||||
Label *FalseTargetLabel;
|
||||
|
||||
auto TrueIter = JumpTargets.find(Op->TrueBlock.ID());
|
||||
auto FalseIter = JumpTargets.find(Op->FalseBlock.ID());
|
||||
|
||||
if (TrueIter == JumpTargets.end()) {
|
||||
TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
}
|
||||
else {
|
||||
TrueTargetLabel = &TrueIter->second;
|
||||
}
|
||||
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
uint64_t Const;
|
||||
@@ -154,24 +149,18 @@ DEF_OP(CondJump) {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4)
|
||||
if (Op->CompareSize == 4) {
|
||||
ucomiss(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
|
||||
else
|
||||
} else {
|
||||
ucomisd(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
auto [_, __, JCC] = GetCC(Op->Cond);
|
||||
|
||||
(this->*JCC)(*TrueTargetLabel, T_NEAR);
|
||||
|
||||
if (FalseIter == JumpTargets.end()) {
|
||||
FalseTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
}
|
||||
else {
|
||||
FalseTargetLabel = &FalseIter->second;
|
||||
}
|
||||
|
||||
PendingTargetLabel = FalseTargetLabel;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
|
||||
@@ -5,11 +5,17 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
movapd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
|
||||
@@ -5,10 +5,15 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
@@ -40,6 +45,36 @@ DEF_OP(AESKeyGenAssist) {
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->RCON);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (IROp->Size) {
|
||||
case 4:
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", IROp->Size);
|
||||
}
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt8());
|
||||
break;
|
||||
case 2:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt16());
|
||||
break;
|
||||
case 4:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt32());
|
||||
break;
|
||||
case 8:
|
||||
crc32(GetDst<RA_64>(Node).cvt64(), TMP1.cvt64());
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -49,7 +84,7 @@ void X86JITCore::RegisterEncryptionHandlers() {
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -5,11 +5,16 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
|
||||
|
||||
+68
-49
@@ -8,27 +8,43 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bits/types/stack_t.h>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <signal.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include <sys/mman.h>
|
||||
#include <tuple>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size) {
|
||||
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
@@ -38,6 +54,9 @@ CodeBuffer AllocateNewCodeBuffer(size_t Size) {
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
@@ -45,10 +64,6 @@ void FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void X86JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
@@ -83,7 +98,7 @@ void X86JITCore::PopRegs() {
|
||||
add(rsp, 16 * RAXMM_x.size());
|
||||
}
|
||||
|
||||
void X86JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -293,7 +308,7 @@ void X86JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
}
|
||||
}
|
||||
|
||||
void X86JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread)
|
||||
@@ -304,7 +319,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
{
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
|
||||
RAPass = Thread->PassManager->GetRAPass();
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, NumGPRs);
|
||||
@@ -342,6 +357,9 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
ThreadSharedData.OverflowExceptionInstructionAddress = Dispatcher->OverflowExceptionInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
@@ -360,7 +378,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
@@ -402,7 +420,7 @@ void X86JITCore::ClearCache() {
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(CTX, CurrentCodeBuffer->Size);
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
@@ -410,13 +428,13 @@ void X86JITCore::ClearCache() {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(X86JITCore::INITIAL_CODE_SIZE);
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(CTX, X86JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
setNewBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(uint32_t Node) const {
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
@@ -424,16 +442,16 @@ IR::PhysicalRegister X86JITCore::GetPhys(uint32_t Node) const {
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsFPR(uint32_t Node) const {
|
||||
bool X86JITCore::IsFPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPR(uint32_t Node) const {
|
||||
bool X86JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg X86JITCore::GetSrc(uint32_t Node) const {
|
||||
Xbyak::Reg X86JITCore::GetSrc(IR::NodeID Node) const {
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
// r10
|
||||
// Callee Saved
|
||||
@@ -452,24 +470,24 @@ Xbyak::Reg X86JITCore::GetSrc(uint32_t Node) const {
|
||||
}
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_64>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_64>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_32>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_32>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_16>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_16>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_8>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_8>(IR::NodeID Node) const;
|
||||
|
||||
Xbyak::Xmm X86JITCore::GetSrc(uint32_t Node) const {
|
||||
Xbyak::Xmm X86JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
return RAXMM_x[PhyReg.Reg];
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg X86JITCore::GetDst(uint32_t Node) const {
|
||||
Xbyak::Reg X86JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if constexpr (RAType == RA_64)
|
||||
return RA64[PhyReg.Reg].cvt64();
|
||||
@@ -484,19 +502,19 @@ Xbyak::Reg X86JITCore::GetDst(uint32_t Node) const {
|
||||
}
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_64>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_64>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_32>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_32>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_16>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_16>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_8>(uint32_t Node) const;
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_8>(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(uint32_t Node) const {
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(IR::NodeID Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if constexpr (RAType == RA_64)
|
||||
return RA64Pair[PhyReg.Reg];
|
||||
@@ -505,12 +523,12 @@ std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(uint32_t Node) const {
|
||||
}
|
||||
|
||||
template
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_64>(uint32_t Node) const;
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_64>(IR::NodeID Node) const;
|
||||
|
||||
template
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_32>(uint32_t Node) const;
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_32>(IR::NodeID Node) const;
|
||||
|
||||
Xbyak::Xmm X86JITCore::GetDst(uint32_t Node) const {
|
||||
Xbyak::Xmm X86JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
return RAXMM_x[PhyReg.Reg];
|
||||
}
|
||||
@@ -535,7 +553,12 @@ bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, u
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = Entry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
@@ -672,15 +695,11 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
|
||||
// if there is a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
{
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
|
||||
jmp(*PendingTargetLabel, T_NEAR);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
@@ -709,10 +728,10 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
Inst << "\t" << Name << " ";
|
||||
}
|
||||
|
||||
uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
uint32_t ArgNode = IROp->Args[i].ID();
|
||||
uint64_t PhysReg = RAPass->GetNodeRegister(ArgNode);
|
||||
const auto ArgNode = IROp->Args[i].ID();
|
||||
const uint64_t PhysReg = RAPass->GetNodeRegister(ArgNode);
|
||||
if (PhysReg >= GPRPairBase)
|
||||
Inst << "Pair" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
else if (PhysReg >= XMMBase)
|
||||
@@ -724,7 +743,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
LogMan::Msg::DFmt("{}", Inst.str());
|
||||
}
|
||||
#endif
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
@@ -772,6 +791,6 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
}
|
||||
}
|
||||
+43
-28
@@ -9,8 +9,6 @@ $end_info$
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include "Common/MathUtils.h"
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
@@ -20,6 +18,7 @@ using namespace Xbyak;
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <tuple>
|
||||
@@ -30,14 +29,9 @@ struct CodeBuffer {
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
// r10, r11
|
||||
@@ -62,14 +56,22 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
|
||||
|
||||
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread);
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
CodeBuffer Buffer,
|
||||
bool CompileThread);
|
||||
~X86JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(uint64_t Entry, FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
|
||||
bool NeedsOpDispatch() override { return true; }
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
@@ -77,6 +79,10 @@ public:
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
}
|
||||
|
||||
private:
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::Context *CTX;
|
||||
@@ -85,7 +91,7 @@ private:
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
uint64_t Entry;
|
||||
|
||||
std::unordered_map<IR::OrderedNodeWrapper::NodeOffsetType, Label> JumpTargets;
|
||||
std::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
Xbyak::util::Cpu Features{};
|
||||
|
||||
bool MemoryDebug = false;
|
||||
@@ -111,26 +117,27 @@ private:
|
||||
constexpr static uint8_t RA_64 = 3;
|
||||
constexpr static uint8_t RA_XMM = 4;
|
||||
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node) const;
|
||||
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const;
|
||||
|
||||
bool IsFPR(uint32_t Node) const;
|
||||
bool IsGPR(uint32_t Node) const;
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg GetSrc(uint32_t Node) const;
|
||||
[[nodiscard]] Xbyak::Reg GetSrc(IR::NodeID Node) const;
|
||||
template<uint8_t RAType>
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> GetSrcPair(uint32_t Node) const;
|
||||
[[nodiscard]] std::pair<Xbyak::Reg, Xbyak::Reg> GetSrcPair(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg GetDst(uint32_t Node) const;
|
||||
[[nodiscard]] Xbyak::Reg GetDst(IR::NodeID Node) const;
|
||||
|
||||
Xbyak::Xmm GetSrc(uint32_t Node) const;
|
||||
Xbyak::Xmm GetDst(uint32_t Node) const;
|
||||
[[nodiscard]] Xbyak::Xmm GetSrc(IR::NodeID Node) const;
|
||||
[[nodiscard]] Xbyak::Xmm GetDst(IR::NodeID Node) const;
|
||||
|
||||
Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const;
|
||||
[[nodiscard]] Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale) const;
|
||||
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
FEXCore::IR::RegisterAllocationData *RAData;
|
||||
@@ -159,6 +166,8 @@ private:
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
@@ -173,8 +182,8 @@ private:
|
||||
|
||||
std::tuple<SetCC, CMovCC, JCC> GetCC(IR::CondClassType cond);
|
||||
|
||||
using OpHandler = void (X86JITCore::*)(FEXCore::IR::IROp_Header *IROp, uint32_t Node);
|
||||
std::array<OpHandler, FEXCore::IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
using OpHandler = void (X86JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
void RegisterBranchHandlers();
|
||||
@@ -188,7 +197,7 @@ private:
|
||||
|
||||
void PushRegs();
|
||||
void PopRegs();
|
||||
#define DEF_OP(x) void Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
@@ -216,6 +225,7 @@ private:
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
@@ -223,6 +233,8 @@ private:
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
@@ -305,6 +317,7 @@ private:
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
@@ -315,6 +328,7 @@ private:
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -420,6 +434,7 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
+52
-28
@@ -5,13 +5,19 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <cmath>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
@@ -136,7 +142,7 @@ DEF_OP(StoreContext) {
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = Op->Size;
|
||||
size_t size = IROp->Size;
|
||||
Reg index = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
@@ -160,7 +166,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -196,7 +202,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -225,7 +231,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
movups(GetDst(Node), xword [STATE + rax]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -240,7 +246,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
Reg index = GetSrc<RA_64>(Op->Header.Args[1].ID());
|
||||
size_t size = Op->Size;
|
||||
size_t size = IROp->Size;
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
auto value = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
@@ -252,9 +258,9 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 4:
|
||||
case 8: {
|
||||
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", Op->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
}
|
||||
mov(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value);
|
||||
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -272,16 +278,16 @@ DEF_OP(StoreContextIndexed) {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
@@ -295,16 +301,16 @@ DEF_OP(StoreContextIndexed) {
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(Op->Size * 8) [STATE + rax], value, 0);
|
||||
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(Op->Size * 8) [STATE + rax], value, 0);
|
||||
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(Op->Size * 8) [STATE + rax], value);
|
||||
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(Op->Size * 8) [STATE + rax], value);
|
||||
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
@@ -466,7 +472,7 @@ DEF_OP(LoadMem) {
|
||||
if (Op->Class.Val == 0) {
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
movzx (Dst, byte [MemPtr]);
|
||||
}
|
||||
@@ -483,14 +489,14 @@ DEF_OP(LoadMem) {
|
||||
mov(Dst, qword [MemPtr]);
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
auto Dst = GetDst(Node);
|
||||
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
movzx(eax, byte [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
@@ -510,7 +516,7 @@ DEF_OP(LoadMem) {
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Size == Op->Align)
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
else
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
@@ -519,7 +525,7 @@ DEF_OP(LoadMem) {
|
||||
}
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -532,7 +538,7 @@ DEF_OP(StoreMem) {
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class.Val == 0) {
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
@@ -545,11 +551,11 @@ DEF_OP(StoreMem) {
|
||||
case 8:
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Op->Size) {
|
||||
switch (IROp->Size) {
|
||||
case 1:
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Header.Args[1].ID()), 0);
|
||||
break;
|
||||
@@ -563,12 +569,12 @@ DEF_OP(StoreMem) {
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 16:
|
||||
if (Op->Size == Op->Align)
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
else
|
||||
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", Op->Size);
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -589,6 +595,23 @@ DEF_OP(CacheLineClear) {
|
||||
clflush(ptr [MemReg]);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
// Align by cacheline
|
||||
mov (TMP1, CPUIDEmu::CACHELINE_SIZE - 1);
|
||||
andn(TMP1, TMP1, MemReg.cvt64());
|
||||
xor_(TMP2, TMP2);
|
||||
|
||||
using DataType = uint64_t;
|
||||
// 64-byte cache line zero
|
||||
for (size_t i = 0; i < CPUIDEmu::CACHELINE_SIZE; i += sizeof(DataType)) {
|
||||
mov (qword [TMP1 + i], TMP2);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -609,6 +632,7 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+40
-14
@@ -4,8 +4,18 @@ tags: backend|x86-64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static void PrintValue(uint64_t Value) {
|
||||
@@ -16,7 +26,7 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
@@ -37,20 +47,16 @@ DEF_OP(Fence) {
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case 0: // Hard fault
|
||||
case 5: // Guest ud2
|
||||
ud2();
|
||||
break;
|
||||
case 1: // Int <imm8>
|
||||
ud2();
|
||||
break;
|
||||
case 2: // overflow
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
ud2();
|
||||
break;
|
||||
case 3: // int 1
|
||||
ud2();
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
|
||||
jmp(TMP1);
|
||||
break;
|
||||
case 4: { // HLT
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
@@ -60,7 +66,7 @@ DEF_OP(Break) {
|
||||
jmp(TMP1);
|
||||
break;
|
||||
}
|
||||
case 6: // INT3
|
||||
case FEXCore::IR::Break_Interrupt3: // INT3
|
||||
{
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Adjust the stack first for a regular return
|
||||
@@ -83,6 +89,18 @@ DEF_OP(Break) {
|
||||
}
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
|
||||
jmp(TMP1);
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
}
|
||||
}
|
||||
@@ -141,6 +159,13 @@ DEF_OP(Print) {
|
||||
PopRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
// Cyclecounter in EDX:EAX
|
||||
// IA32_TSC_AUX in ECX
|
||||
rdtscp();
|
||||
mov (GetDst<RA_32>(Node), ecx);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -157,6 +182,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,11 +5,17 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
|
||||
@@ -5,12 +5,18 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VectorZero) {
|
||||
auto Dst = GetDst(Node);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
+6
-5
@@ -5,10 +5,11 @@ desc: Stores information about blocks, and provides C++ implementations to looku
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
|
||||
@@ -36,11 +37,11 @@ LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
|
||||
+10
-2
@@ -1,10 +1,18 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
@@ -42,7 +50,7 @@ public:
|
||||
auto InsertPoint =
|
||||
#endif
|
||||
BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LOGMAN_THROW_A(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
LOGMAN_THROW_A_FMT(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
CodePages[CurrentPage].push_back(Address);
|
||||
|
||||
+1053
-601
File diff suppressed because it is too large.
Load diff
+634
-51
@@ -3,7 +3,9 @@
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
@@ -12,9 +14,11 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <fmt/format.h>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
@@ -24,18 +28,51 @@ class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
enum {
|
||||
FLAGS_OP_NONE, // must rely on x86 flags
|
||||
FLAGS_OP_CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
FLAGS_OP_AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FLAGS_OP_FCMP, // flags were set by a ucomis* / comis*
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
public:
|
||||
int flagsOp;
|
||||
uint8_t flagsOpSize;
|
||||
OrderedNode* flagsOpDest, *flagsOpSrc;
|
||||
OrderedNode* flagsOpDestSigned, *flagsOpSrcSigned;
|
||||
enum class FlagsGenerationType : uint8_t {
|
||||
TYPE_NONE,
|
||||
TYPE_ADC,
|
||||
TYPE_SBB,
|
||||
TYPE_SUB,
|
||||
TYPE_ADD,
|
||||
TYPE_MUL,
|
||||
TYPE_UMUL,
|
||||
TYPE_LOGICAL,
|
||||
TYPE_LSHL,
|
||||
TYPE_LSHLI,
|
||||
TYPE_LSHR,
|
||||
TYPE_LSHRI,
|
||||
TYPE_ASHR,
|
||||
TYPE_ASHRI,
|
||||
TYPE_ROR,
|
||||
TYPE_RORI,
|
||||
TYPE_ROL,
|
||||
TYPE_ROLI,
|
||||
TYPE_FCMP,
|
||||
TYPE_BEXTR,
|
||||
TYPE_BLSI,
|
||||
TYPE_BLSMSK,
|
||||
TYPE_BLSR,
|
||||
TYPE_POPCOUNT,
|
||||
TYPE_BZHI,
|
||||
TYPE_TZCNT,
|
||||
TYPE_LZCNT,
|
||||
TYPE_BITSELECT,
|
||||
};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::Context *CTX{};
|
||||
bool ShouldDump {false};
|
||||
@@ -49,7 +86,7 @@ public:
|
||||
|
||||
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
LOGMAN_THROW_A(it != JumpTargets.end(), "Couldn't find block generated for 0x%lx", RIP);
|
||||
LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP);
|
||||
return it->second.BlockEntry;
|
||||
}
|
||||
|
||||
@@ -67,7 +104,7 @@ public:
|
||||
}
|
||||
|
||||
void StartNewBlock() {
|
||||
flagsOp = FLAGS_OP_NONE;
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
@@ -82,6 +119,9 @@ public:
|
||||
// cmp qword [rdi-8], 0
|
||||
// jne .label
|
||||
if (LastOp && !BlockSetRIP) {
|
||||
// Calculate flags first
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto it = JumpTargets.find(NextRIP);
|
||||
if (it == JumpTargets.end()) {
|
||||
|
||||
@@ -96,6 +136,11 @@ public:
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (LastOp) {
|
||||
LOGMAN_THROW_A_FMT(IsDeferredFlagsStored(), "FinishOp: Deferred flags weren't generated at end of block");
|
||||
}
|
||||
|
||||
BlockSetRIP = false;
|
||||
|
||||
return false;
|
||||
@@ -229,9 +274,9 @@ public:
|
||||
void PopcountOp(OpcodeArgs);
|
||||
void XLATOp(OpcodeArgs);
|
||||
|
||||
enum Segment {
|
||||
Segment_FS,
|
||||
Segment_GS,
|
||||
enum class Segment {
|
||||
FS,
|
||||
GS,
|
||||
};
|
||||
template<Segment Seg>
|
||||
void ReadSegmentReg(OpcodeArgs);
|
||||
@@ -322,6 +367,24 @@ public:
|
||||
template<size_t ElementSize>
|
||||
void PSIGN(OpcodeArgs);
|
||||
|
||||
// BMI1 Ops
|
||||
void ANDNBMIOp(OpcodeArgs);
|
||||
void BEXTRBMIOp(OpcodeArgs);
|
||||
void BLSIBMIOp(OpcodeArgs);
|
||||
void BLSMSKBMIOp(OpcodeArgs);
|
||||
void BLSRBMIOp(OpcodeArgs);
|
||||
|
||||
// BMI2 Ops
|
||||
void BMI2Shift(OpcodeArgs);
|
||||
void BZHI(OpcodeArgs);
|
||||
void MULX(OpcodeArgs);
|
||||
void PDEP(OpcodeArgs);
|
||||
void PEXT(OpcodeArgs);
|
||||
void RORX(OpcodeArgs);
|
||||
|
||||
// ADX Ops
|
||||
void ADXOp(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
template<size_t width>
|
||||
void FLD(OpcodeArgs);
|
||||
@@ -448,6 +511,8 @@ public:
|
||||
void FenceOp(OpcodeArgs);
|
||||
|
||||
void StoreFenceOrCLFlush(OpcodeArgs);
|
||||
void CLZeroOp(OpcodeArgs);
|
||||
void RDTSCPOp(OpcodeArgs);
|
||||
|
||||
void PSADBW(OpcodeArgs);
|
||||
|
||||
@@ -475,8 +540,12 @@ public:
|
||||
|
||||
void MPSADBWOp(OpcodeArgs);
|
||||
|
||||
void CRC32(OpcodeArgs);
|
||||
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
|
||||
@@ -492,24 +561,35 @@ private:
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
|
||||
OrderedNode *GetDynamicPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align);
|
||||
|
||||
uint8_t GetDstSize(FEXCore::X86Tables::DecodedOp Op) const;
|
||||
uint8_t GetSrcSize(FEXCore::X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
|
||||
}
|
||||
|
||||
[[nodiscard]] static uint32_t MMBaseOffset() {
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
|
||||
[[nodiscard]] uint8_t GetDstSize(X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] uint8_t GetSrcSize(X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] uint32_t GetDstBitSize(X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] uint32_t GetSrcBitSize(X86Tables::DecodedOp Op) const;
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value) {
|
||||
flagsOp = FLAGS_OP_NONE;
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
|
||||
flagsOp = FLAGS_OP_NONE;
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
}
|
||||
|
||||
@@ -519,32 +599,524 @@ private:
|
||||
|
||||
OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
|
||||
void GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High);
|
||||
void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High);
|
||||
void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
/**
|
||||
* @name Deferred RFLAG calculation and generation.
|
||||
*
|
||||
* Only handles the six flags that ALU ops typically generate.
|
||||
* Specifically: CF, PF, AF, ZF, SF, OF
|
||||
* These six flags are heavily generated through basic ALU ops and balloon the IR if not early eliminated.
|
||||
* This tracking structure only tracks single blocks and requires RFLAGS calculation at block-ending ops.
|
||||
* Some flags generating ALU ops only touch part of the registers, In these cases it will do calculation up front.
|
||||
* This means we still need our IR passes to eliminate all redundant flags accesses but this light OpcodeDispatcher optimization
|
||||
* doesn't take it to that level.
|
||||
* @{ */
|
||||
|
||||
// Deferred flag generation tracking structure.
|
||||
// This structure is used to track RFlags from ALU ops for invalidation.
|
||||
//
|
||||
// Future ideas: Use an invalidation mask to do partial generation of flags.
|
||||
// Particularly for the instructions that don't do the full set of flags calculations.
|
||||
// These instructions currently calculate the deferred RFLAGS immediately then overwrite rflags state.
|
||||
// RCLSE IR pass will catch and remove redundant rflags stores like this currently.
|
||||
struct DeferredFlagData {
|
||||
// What type of flags to generate
|
||||
FlagsGenerationType Type {FlagsGenerationType::TYPE_NONE};
|
||||
|
||||
// Source size of the op
|
||||
uint8_t SrcSize;
|
||||
|
||||
// Every flag generation type has a result
|
||||
OrderedNode *Res{};
|
||||
|
||||
union {
|
||||
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT
|
||||
struct {
|
||||
} NoSource;
|
||||
|
||||
// MUL, BLSR, BZHI
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
} OneSource;
|
||||
|
||||
// Logical, LSHL, LSHR, ASHR, ROR, ROL
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
OrderedNode *Src2;
|
||||
} TwoSource;
|
||||
|
||||
// ADC, SBB
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
OrderedNode *Src2;
|
||||
OrderedNode *Src3;
|
||||
} ThreeSource;
|
||||
|
||||
// LSHLI, LSHRI, ASHRI, RORI, ROLI
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
uint64_t Imm;
|
||||
} OneSrcImmediate;
|
||||
|
||||
// ADD, SUB
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
OrderedNode *Src2;
|
||||
|
||||
bool UpdateCF;
|
||||
} TwoSrcImmediate;
|
||||
} Sources{};
|
||||
};
|
||||
|
||||
DeferredFlagData CurrentDeferredFlags{};
|
||||
|
||||
/**
|
||||
* @brief Takes the current deferred flag state and stores the result in to RFLAGS.
|
||||
*
|
||||
* Once executed there will no longer be any deferred flag state and RFLAGS will have the correct flags in it.
|
||||
* Necessary to do when leaving a IR block, or if an instruction is doing a partial overwrite of the flags.
|
||||
*/
|
||||
void CalculateDeferredFlags(uint32_t FlagsToCalculateMask = ~0U);
|
||||
|
||||
/**
|
||||
* @brief Invalidates the current deferred flags structure.
|
||||
*
|
||||
* If the emulated instruction is going to overwrite all of the flags but isn't tracked using the deferred flag system
|
||||
* then use this function to stop tracking the current active deferred flags.
|
||||
*/
|
||||
void InvalidateDeferredFlags() {
|
||||
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Checks if there is any deferred flag state active.
|
||||
*
|
||||
* @return True if RFLAGs contains the flags. False if deferred flags is tracking the data.
|
||||
*/
|
||||
bool IsDeferredFlagsStored() const {
|
||||
return CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
void CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculcateFlags_UMUL(OrderedNode *High);
|
||||
void CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculcateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculcateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BITSELECT(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name These functions generated deferred RFLAGs tracking.
|
||||
*
|
||||
* Depending on the operation it may force a RFLAGs calculation before storing the new deferred state.
|
||||
* @{ */
|
||||
void GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ADC,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.ThreeSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.Src3 = CF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_SBB,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.ThreeSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.Src3 = CF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
|
||||
if (!UpdateCF) {
|
||||
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_SUB,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.UpdateCF = UpdateCF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
|
||||
if (!UpdateCF) {
|
||||
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ADD,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.UpdateCF = UpdateCF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_MUL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSource = {
|
||||
.Src1 = High,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_UMUL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = High,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LOGICAL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Flags need to be used, generate incoming flags first.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Flags need to be used, generate incoming flags first.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Flags need to be used, generate incoming flags first.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ASHR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHLI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ASHRI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHRI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ROR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ROL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_RORI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ROLI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_FCMP(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_FCMP,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BEXTR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSI(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSMSK,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSource = {
|
||||
.Src1 = Src,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_POPCOUNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_POPCOUNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BZHI(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Result, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BZHI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Result,
|
||||
.Sources = {
|
||||
.OneSource = {
|
||||
.Src1 = Src,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_TZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_TZCNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_LZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LZCNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BITSELECT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BITSELECT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
/** @} */
|
||||
/** @} */
|
||||
|
||||
OrderedNode * GetX87Top();
|
||||
enum X87Tag {
|
||||
TAG_VALID = 0b00,
|
||||
TAG_ZERO = 0b01,
|
||||
TAG_SPECIAL = 0b10,
|
||||
TAG_EMPTY = 0b11
|
||||
enum class X87Tag {
|
||||
Valid = 0b00,
|
||||
Zero = 0b01,
|
||||
Special = 0b10,
|
||||
Empty = 0b11
|
||||
};
|
||||
void SetX87TopTag(OrderedNode *Value, uint32_t Tag);
|
||||
void SetX87TopTag(OrderedNode *Value, X87Tag Tag);
|
||||
OrderedNode *GetX87FTW(OrderedNode *Value);
|
||||
void SetX87Top(OrderedNode *Value);
|
||||
|
||||
@@ -564,22 +1136,33 @@ private:
|
||||
|
||||
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Align = 1) {
|
||||
if (CTX->Config.TSOEnabled)
|
||||
return _StoreMemTSO(ssa0, ssa1, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
|
||||
return _StoreMemTSO(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
|
||||
else
|
||||
return _StoreMem(ssa0, ssa1, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
|
||||
return _StoreMem(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
|
||||
}
|
||||
|
||||
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
if (CTX->Config.TSOEnabled)
|
||||
return _LoadMemTSO(ssa0, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
|
||||
return _LoadMemTSO(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
|
||||
else
|
||||
return _LoadMem(ssa0, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
|
||||
return _LoadMem(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
|
||||
}
|
||||
|
||||
|
||||
};
|
||||
|
||||
void InstallOpcodeHandlers(Context::OperatingMode Mode);
|
||||
|
||||
}
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::OpDispatchBuilder::FlagsGenerationType> : fmt::formatter<int> {
|
||||
using Base = fmt::formatter<int>;
|
||||
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::OpDispatchBuilder::FlagsGenerationType& ID, FormatContext& ctx) {
|
||||
return Base::format(static_cast<int>(ID), ctx);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -5,11 +5,16 @@ desc: Handles x86/64 Crypto instructions to IR
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
@@ -48,7 +53,7 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
auto Res = _VAESKeyGenAssist(Src, RCON);
|
||||
|
||||
+506
-54
@@ -5,9 +5,17 @@ desc: Handles x86/64 flag generation
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
@@ -31,35 +39,226 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
};
|
||||
|
||||
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
uint8_t NumFlags = FlagOffsets.size();
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
// Calculate flags early.
|
||||
// Could use InvalidateDeferredFlags() if we had masked invalidation.
|
||||
// This is only a partial overwrite of flags since OF isn't stored here.
|
||||
CalculateDeferredFlags();
|
||||
NumFlags = 5;
|
||||
}
|
||||
else {
|
||||
// We are overwriting all RFLAGS. Invalidate the deferred flag state.
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
for (int i = 0; i < NumFlags; ++i) {
|
||||
auto Tmp = _And(_Lshr(Src, _Constant(FlagOffsets[i])), OneConst);
|
||||
SetRFLAG(Tmp, FlagOffsets[i]);
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
auto Tmp = _And(_Lshr(Src, _Constant(FlagOffset)), OneConst);
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
OrderedNode *Original = _Constant(2);
|
||||
uint8_t NumFlags = FlagOffsets.size();
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
NumFlags = 5;
|
||||
}
|
||||
|
||||
for (int i = 0; i < NumFlags; ++i) {
|
||||
OrderedNode *Flag = _LoadFlag(FlagOffsets[i]);
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
OrderedNode *Flag = _LoadFlag(FlagOffset);
|
||||
Flag = _Bfe(4, 32, 0, Flag);
|
||||
Flag = _Lshl(Flag, _Constant(FlagOffsets[i]));
|
||||
Flag = _Lshl(Flag, _Constant(FlagOffset));
|
||||
Original = _Or(Original, Flag);
|
||||
}
|
||||
return Original;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
auto Size = GetSrcSize(Op) * 8;
|
||||
void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
if (CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE) {
|
||||
// Nothing to do
|
||||
return;
|
||||
}
|
||||
|
||||
switch (CurrentDeferredFlags.Type) {
|
||||
case FlagsGenerationType::TYPE_ADC:
|
||||
CalculcateFlags_ADC(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src1,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src2,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src3);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_SBB:
|
||||
CalculcateFlags_SBB(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src1,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src2,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src3);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_SUB:
|
||||
CalculcateFlags_SUB(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ADD:
|
||||
CalculcateFlags_ADD(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_MUL:
|
||||
CalculcateFlags_MUL(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_UMUL:
|
||||
CalculcateFlags_UMUL(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LOGICAL:
|
||||
CalculcateFlags_Logical(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHL:
|
||||
CalculcateFlags_ShiftLeft(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHLI:
|
||||
CalculcateFlags_ShiftLeftImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHR:
|
||||
CalculcateFlags_ShiftRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHRI:
|
||||
CalculcateFlags_ShiftRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHR:
|
||||
CalculcateFlags_SignShiftRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHRI:
|
||||
CalculcateFlags_SignShiftRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROR:
|
||||
CalculcateFlags_RotateRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_RORI:
|
||||
CalculcateFlags_RotateRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROL:
|
||||
CalculcateFlags_RotateLeft(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROLI:
|
||||
CalculcateFlags_RotateLeftImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_FCMP:
|
||||
CalculcateFlags_FCMP(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BEXTR:
|
||||
CalculcateFlags_BEXTR(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSI:
|
||||
CalculcateFlags_BLSI(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSMSK:
|
||||
CalculcateFlags_BLSMSK(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSR:
|
||||
CalculcateFlags_BLSR(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_POPCOUNT:
|
||||
CalculcateFlags_POPCOUNT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BZHI:
|
||||
CalculcateFlags_BZHI(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_TZCNT:
|
||||
CalculcateFlags_TZCNT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LZCNT:
|
||||
CalculcateFlags_LZCNT(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BITSELECT:
|
||||
CalculcateFlags_BITSELECT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_NONE:
|
||||
default: ERROR_AND_DIE_FMT("Unhandled flags type {}", CurrentDeferredFlags.Type);
|
||||
}
|
||||
|
||||
// Done calculating
|
||||
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
auto Size = SrcSize * 8;
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -69,7 +268,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(Size - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -122,13 +321,15 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
case 64:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown BFESize: %d", Size); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", Size);
|
||||
break;
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -138,7 +339,7 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -177,7 +378,7 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (GetSrcSize(Op)) {
|
||||
switch (SrcSize) {
|
||||
case 1:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
@@ -190,13 +391,15 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown BFESize: %d", GetSrcSize(Op)); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", SrcSize);
|
||||
break;
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -206,7 +409,7 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -247,13 +450,13 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *FinalAnd = _And(XorOp1, XorOp2);
|
||||
|
||||
FinalAnd = _Bfe(1, GetSrcSize(Op) * 8 - 1, FinalAnd);
|
||||
FinalAnd = _Bfe(1, SrcSize * 8 - 1, FinalAnd);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(FinalAnd);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -263,7 +466,7 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -300,7 +503,7 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (GetSrcSize(Op)) {
|
||||
switch (SrcSize) {
|
||||
case 1:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
@@ -313,13 +516,15 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown BFESize: %d", GetSrcSize(Op)); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", SrcSize);
|
||||
break;
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
|
||||
// PF/AF/ZF/SF
|
||||
// Undefined
|
||||
{
|
||||
@@ -334,7 +539,7 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
|
||||
auto SignBit = _Sbfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto SignBit = _Sbfe(1, SrcSize * 8 - 1, Res);
|
||||
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, _Constant(0), _Constant(1));
|
||||
|
||||
@@ -343,7 +548,7 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculcateFlags_UMUL(OrderedNode *High) {
|
||||
// AF/SF/PF/ZF
|
||||
// Undefined
|
||||
{
|
||||
@@ -365,7 +570,7 @@ void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ord
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// AF
|
||||
{
|
||||
// Undefined
|
||||
@@ -375,7 +580,7 @@ void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op,
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -410,11 +615,11 @@ auto oldflag = GetRFLAG(FEXCore::X86State::flag);\
|
||||
auto newval = _Select(FEXCore::IR::COND_EQ, cond, _Constant(0), oldflag, newflag);\
|
||||
SetRFLAG<FEXCore::X86State::flag>(newval);
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
auto Size = _Constant(GetSrcSize(Op) * 8);
|
||||
auto Size = _Constant(SrcSize * 8);
|
||||
auto ShiftAmt = _Sub(Size, Src2);
|
||||
auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1));
|
||||
COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit);
|
||||
@@ -446,7 +651,7 @@ void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
// SF
|
||||
{
|
||||
auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto val = _Bfe(1, SrcSize * 8 - 1, Res);
|
||||
COND_FLAG_SET(Src2, RFLAG_SF_LOC, val);
|
||||
}
|
||||
|
||||
@@ -454,12 +659,12 @@ void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op
|
||||
{
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
// When Shift > 1 then OF is undefined
|
||||
auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res));
|
||||
auto val = _Bfe(1, SrcSize * 8 - 1, _Xor(Src1, Res));
|
||||
COND_FLAG_SET(Src2, RFLAG_OF_LOC, val);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
@@ -494,7 +699,7 @@ void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp O
|
||||
|
||||
// SF
|
||||
{
|
||||
auto val =_Bfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto val =_Bfe(1, SrcSize * 8 - 1, Res);
|
||||
COND_FLAG_SET(Src2, RFLAG_SF_LOC, val);
|
||||
}
|
||||
|
||||
@@ -502,12 +707,12 @@ void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp O
|
||||
{
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// OF flag is set if a sign change occurred
|
||||
auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res));
|
||||
auto val = _Bfe(1, SrcSize * 8 - 1, _Xor(Src1, Res));
|
||||
COND_FLAG_SET(Src2, RFLAG_OF_LOC, val);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
@@ -542,7 +747,7 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::Decoded
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
COND_FLAG_SET(Src2, RFLAG_SF_LOC, LshrOp);
|
||||
@@ -554,14 +759,14 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::Decoded
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, GetSrcSize(Op) * 8 - Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
|
||||
}
|
||||
|
||||
// PF
|
||||
@@ -590,20 +795,20 @@ void OpDispatchBuilder::GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::Dec
|
||||
|
||||
// SF
|
||||
{
|
||||
auto LshrOp = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto LshrOp = _Bfe(1, SrcSize * 8 - 1, Res);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
|
||||
// OF
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
if (Shift == 1) {
|
||||
auto SourceBit = _Bfe(1, GetSrcSize(Op) * 8 - 1, Src1);
|
||||
auto SourceBit = _Bfe(1, SrcSize * 8 - 1, Src1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(SourceBit, LshrOp));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
@@ -639,7 +844,7 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -654,7 +859,7 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
@@ -690,7 +895,7 @@ void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::De
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -701,13 +906,13 @@ void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::De
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// Is set to the MSB of the original value
|
||||
if (Shift == 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Bfe(1, GetSrcSize(Op) * 8 - 1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Bfe(1, SrcSize * 8 - 1, Src1));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
auto NewCF = _Bfe(1, OpSize - 1, Res);
|
||||
@@ -735,8 +940,8 @@ void OpDispatchBuilder::GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
//auto Size = _Constant(GetSrcSize(Res) * 8);
|
||||
@@ -765,10 +970,10 @@ void OpDispatchBuilder::GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp O
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
|
||||
|
||||
@@ -787,10 +992,10 @@ void OpDispatchBuilder::GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::D
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -807,4 +1012,251 @@ void OpDispatchBuilder::GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::De
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(HostFlag_Unordered);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(ZeroConst);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BEXTR(OrderedNode *Src) {
|
||||
// Handle flag setting.
|
||||
//
|
||||
// All that matters primarily for this instruction is
|
||||
// that we only set the ZF flag properly.
|
||||
//
|
||||
// CF and OF are defined as being set to zero
|
||||
//
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(_Constant(0));
|
||||
|
||||
// Every other flag is considered undefined after a
|
||||
// BEXTR instruction, but we opt to reliably clear them.
|
||||
//
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(_Constant(0));
|
||||
|
||||
// PF
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// ZF
|
||||
auto ZeroOp = _Select(IR::COND_EQ,
|
||||
Src, _Constant(0),
|
||||
_Constant(1), _Constant(0));
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZeroOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src) {
|
||||
// Now for the flags:
|
||||
//
|
||||
// Only CF, SF, ZF and OF are defined as being updated
|
||||
// CF is cleared if Src is zero, otherwise it's set.
|
||||
// SF is set to the value of the most significant operand bit of Result.
|
||||
// OF is always cleared
|
||||
// ZF is set, as usual, if Result is zero or not.
|
||||
//
|
||||
// AF and PF are documented as being in an undefined state after
|
||||
// a BLSI operation, however, we choose to reliably clear them.
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
Zero, One);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBit = _Constant(SrcSize * 8 - 1);
|
||||
auto SFOp = _Lshr(Src, SignBit);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(SFOp);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BLSMSK(OrderedNode *Src) {
|
||||
// Now for the flags.
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
auto CFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
Zero, One);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
// Now for flags.
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Result, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
Zero, One);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBit = _Constant(SrcSize * 8 - 1);
|
||||
auto SFOp = _Lshr(Result, SignBit);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(SFOp);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_POPCOUNT(OrderedNode *Src) {
|
||||
// Set ZF
|
||||
auto Zero = _Constant(0);
|
||||
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, Zero,
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(Zero);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
// Now for the flags
|
||||
|
||||
auto Bounds = _Constant(SrcSize * 8- 1);
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Result, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_UGT,
|
||||
Src, Bounds,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SFOp = _Lshr(Result, Bounds);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(SFOp);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_TZCNT(OrderedNode *Src) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
auto Zero = _Constant(0);
|
||||
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, Zero,
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Bfe(1, 0, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, Zero,
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Bfe(1, SrcSize * 8 - 1, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BITSELECT(OrderedNode *Src) {
|
||||
// OF, SF, AF, PF, CF all undefined
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
// ZF is set to 1 if the source was zero
|
||||
auto ZFSelectOp = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, ZeroConst,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFSelectOp);
|
||||
}
|
||||
|
||||
}
|
||||
Loaded 100 of 518 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user