mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 20:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
663fd5a98b | ||
|
|
93e58bc15e | ||
|
|
3ccdf6508e | ||
|
|
fd33cf1ce5 | ||
|
|
2bb64ad1c6 | ||
|
|
e3de62058b | ||
|
|
6ee9984280 | ||
|
|
e1df548ae9 | ||
|
|
79a685c15e | ||
|
|
a61ab2803c | ||
|
|
209ad27332 | ||
|
|
df08981475 | ||
|
|
5ce6039a02 | ||
|
|
6c3fdf723a | ||
|
|
1a617c1eb2 | ||
|
|
3a9b801400 | ||
|
|
6e663acdad | ||
|
|
ae1023bb7a | ||
|
|
9cf25e276d | ||
|
|
8db3670ecc | ||
|
|
baee367532 | ||
|
|
8da4e72d87 | ||
|
|
b4a84a2317 | ||
|
|
43d6347212 | ||
|
|
438501e49c | ||
|
|
c034e99aaf | ||
|
|
d2d0ca2de9 | ||
|
|
cbe2b442b2 | ||
|
|
c5b1cd6e7d | ||
|
|
8d20d1dae3 | ||
|
|
1f2d702c4e | ||
|
|
d2d35d0553 | ||
|
|
1e7f54dd7e | ||
|
|
8430a2f7e6 | ||
|
|
326cde78e6 | ||
|
|
bbc2b0b42f | ||
|
|
231a2c54aa | ||
|
|
36ae4cee73 | ||
|
|
62e5ee2201 | ||
|
|
b89ebd931e | ||
|
|
d1b4ddaf61 | ||
|
|
734a0b236b | ||
|
|
2229c04d4d | ||
|
|
07cff27fa2 | ||
|
|
2cf86998bc | ||
|
|
6ed15a6fd6 | ||
|
|
7610243b0c | ||
|
|
e11349b577 | ||
|
|
9b425697cb | ||
|
|
42596ff91e | ||
|
|
b3c2ff47f3 | ||
|
|
75a0bc79be | ||
|
|
5a002ad08d | ||
|
|
fbac6f86d1 | ||
|
|
ca18bf2a3d | ||
|
|
199effdff7 | ||
|
|
8212f4b7fb | ||
|
|
e1a45a2720 | ||
|
|
bd7edd8651 | ||
|
|
4e1d10a46f | ||
|
|
70b6bc2bae | ||
|
|
46d019fe02 | ||
|
|
d56f689e15 | ||
|
|
90702b4102 | ||
|
|
0cf105b64a | ||
|
|
5c74d9458c | ||
|
|
75391bf834 | ||
|
|
63304a1d88 | ||
|
|
1ab79bd72e | ||
|
|
985bdf2b6c | ||
|
|
b57ea83aea | ||
|
|
d716e22476 | ||
|
|
bbb8e1ccab | ||
|
|
96f20779c6 | ||
|
|
908313e378 | ||
|
|
12e5c60633 | ||
|
|
11946ffc4c | ||
|
|
f44cd9c545 | ||
|
|
9c5ccb13de | ||
|
|
e9bcfd4784 | ||
|
|
fd1e8d4566 | ||
|
|
68dcce0739 | ||
|
|
3c554cd787 | ||
|
|
9e5f2269d9 | ||
|
|
81fc502c6c | ||
|
|
eda8ca5449 | ||
|
|
35dd8972e2 | ||
|
|
643dd56b74 | ||
|
|
b748eab4ed | ||
|
|
f4eaab6977 | ||
|
|
900c114831 | ||
|
|
7937b7e52d | ||
|
|
b40e707771 | ||
|
|
edde5c8516 | ||
|
|
197facb845 | ||
|
|
ca697d0d5d | ||
|
|
32ddf790d5 | ||
|
|
2cfba8c6d0 | ||
|
|
cf6c0765fa | ||
|
|
ad93f27271 | ||
|
|
92428e5cbc | ||
|
|
02c00a87dd | ||
|
|
6c77bc12c1 | ||
|
|
fbb428c249 | ||
|
|
854a741ea4 | ||
|
|
ed1952a79a | ||
|
|
39e8f5122f | ||
|
|
6ee77eb1a1 | ||
|
|
46fc45b952 | ||
|
|
79d90f3c7f | ||
|
|
eb41cb2261 | ||
|
|
87e8a9b6aa | ||
|
|
69f5aa8e35 | ||
|
|
d654f55c3c | ||
|
|
eb1689e79f | ||
|
|
cf6472df90 | ||
|
|
bc6295a78d | ||
|
|
ed5774b88f | ||
|
|
40a29ca9a7 | ||
|
|
a1e1838b11 | ||
|
|
4731dbab2f | ||
|
|
f2841ccb5e | ||
|
|
81a474fe12 | ||
|
|
5af2477d00 | ||
|
|
4cbba94a29 | ||
|
|
2a57428314 | ||
|
|
f9ac57205b | ||
|
|
3c6246f99a | ||
|
|
072e7bd241 | ||
|
|
4080dca816 | ||
|
|
c22fb80129 | ||
|
|
123fcd4bea | ||
|
|
e0c17672f2 | ||
|
|
53d484c5f4 | ||
|
|
a2ea767b77 | ||
|
|
4bd51a4be0 | ||
|
|
474f2dc267 | ||
|
|
033310234e | ||
|
|
10835c4b62 | ||
|
|
3e741df77b | ||
|
|
31d4b4a201 | ||
|
|
678985cbf5 | ||
|
|
e83e9cb6b3 | ||
|
|
48a51679b8 | ||
|
|
987fc925ee | ||
|
|
b6cbb23781 | ||
|
|
cbb9017c12 | ||
|
|
832d1e8d2c | ||
|
|
34af7f942b | ||
|
|
5f390c16be | ||
|
|
f36ac0498d | ||
|
|
18bdd9b665 | ||
|
|
5fd3852fd3 | ||
|
|
764432c77c | ||
|
|
55c43229ca | ||
|
|
dbc4daaaa4 | ||
|
|
acdbbec78d | ||
|
|
3247e777c0 | ||
|
|
c6e60ff3f5 | ||
|
|
b7df102919 | ||
|
|
f5d450b95c | ||
|
|
972aaf9f5f | ||
|
|
b98d5f30c0 | ||
|
|
d2e2e189d8 | ||
|
|
17d110575c | ||
|
|
a967976420 | ||
|
|
7e1e1ccccc | ||
|
|
502452602e | ||
|
|
f8ff46f3e3 | ||
|
|
8f572deaed | ||
|
|
601dd1d2b5 | ||
|
|
673e826e46 | ||
|
|
f1f81f9de2 | ||
|
|
8513c02ec1 | ||
|
|
320c5f1847 | ||
|
|
282f091e85 | ||
|
|
8647033029 | ||
|
|
3b88947cbe | ||
|
|
f05be0120e | ||
|
|
10c3b2fb5d | ||
|
|
74ae89865c | ||
|
|
a2445904e7 | ||
|
|
bd5188e7bc | ||
|
|
d89c119d9d | ||
|
|
802f3bed47 | ||
|
|
9c606a278d | ||
|
|
7aa5bc0503 | ||
|
|
d94b9fae94 | ||
|
|
05fcaaa758 | ||
|
|
9626a64340 | ||
|
|
c1cfd4db83 | ||
|
|
bfda91ac16 | ||
|
|
b7117b86ea | ||
|
|
a94059ad81 | ||
|
|
2cab87fbd0 | ||
|
|
2017100ff3 | ||
|
|
72cb29d36b | ||
|
|
a2b0d594fb | ||
|
|
a16d4ff1f3 | ||
|
|
305b1ecf2d | ||
|
|
3cc0cae249 | ||
|
|
137aa59254 | ||
|
|
5f8cb0dc54 | ||
|
|
45978474f3 | ||
|
|
75206c7a51 | ||
|
|
06ff0a45a7 | ||
|
|
c948d532a7 | ||
|
|
3825483f79 | ||
|
|
0b238d942e | ||
|
|
16b09a9dd3 | ||
|
|
4f6800b768 | ||
|
|
c84801adb3 | ||
|
|
69f990c1f8 | ||
|
|
8a83564678 | ||
|
|
43d93f836f | ||
|
|
7c2c0f7fe5 | ||
|
|
3ff47cf677 | ||
|
|
3165dce89e | ||
|
|
d4fad0a3b0 | ||
|
|
15fcaa794f | ||
|
|
6a1a505336 | ||
|
|
85937a7bfb | ||
|
|
3baa598b9e | ||
|
|
6f83b10af6 | ||
|
|
f52bcb49ab | ||
|
|
9cef8ff7ce | ||
|
|
a499ad6404 | ||
|
|
7c8767ab32 | ||
|
|
38a8559943 | ||
|
|
48c6acbc87 | ||
|
|
b5d93bc4a0 | ||
|
|
a9fffe771b | ||
|
|
69f3ab14b9 | ||
|
|
a7b9a34ec9 | ||
|
|
f45be1f59e | ||
|
|
012c2ba851 | ||
|
|
a0c2ce0068 | ||
|
|
86898035da | ||
|
|
05eeffe0f7 | ||
|
|
70cffce5eb | ||
|
|
0258e914e0 | ||
|
|
c35b81d2c5 | ||
|
|
fc63dcedb6 | ||
|
|
c879650c4b | ||
|
|
c4d019939b | ||
|
|
3841cc4aa5 | ||
|
|
468f2f2d2d | ||
|
|
be5db91c13 | ||
|
|
6e7d0a520a | ||
|
|
e37996873a | ||
|
|
25e851b570 | ||
|
|
d41b21a626 | ||
|
|
d65c54ba21 | ||
|
|
a5d3bf0f48 | ||
|
|
f76e8c7185 | ||
|
|
96145e8e3f | ||
|
|
530821fc4a | ||
|
|
e4a4a529dd | ||
|
|
863d1e0007 | ||
|
|
c64d6f2b36 | ||
|
|
a798880ac8 | ||
|
|
9db9be9612 | ||
|
|
6a19c77184 | ||
|
|
c17b9dea30 | ||
|
|
7564b6c69a | ||
|
|
5858c5041f | ||
|
|
18d0dec416 | ||
|
|
752046108c | ||
|
|
773ddbf5c4 | ||
|
|
88e90a6f3a | ||
|
|
84325d6b5c | ||
|
|
1b8e03e1de | ||
|
|
569d7297f0 | ||
|
|
d9520c9498 | ||
|
|
6b61093037 | ||
|
|
74b09d5581 | ||
|
|
08da8f6f5c | ||
|
|
7f8fffbb24 | ||
|
|
b08e5f9821 | ||
|
|
1758648f56 | ||
|
|
418ce0aed9 | ||
|
|
4e926076c9 | ||
|
|
3f0aaeba68 | ||
|
|
6ecec77a74 | ||
|
|
79917dc9bb | ||
|
|
b0b60fb761 | ||
|
|
7463152f50 | ||
|
|
cbd8217c60 | ||
|
|
4de1762c21 | ||
|
|
73962402df | ||
|
|
d90ccceec7 | ||
|
|
0c5ef1ce1d | ||
|
|
e5680e031d | ||
|
|
eee0cffa57 | ||
|
|
c480b0ba41 | ||
|
|
e7a47a647c | ||
|
|
500c82bbf8 | ||
|
|
008528af91 | ||
|
|
13199ad203 | ||
|
|
533dd1a54b | ||
|
|
b6f2e10416 | ||
|
|
c58db36a7c | ||
|
|
4593099882 | ||
|
|
21e29af48b | ||
|
|
892e44a900 | ||
|
|
bca29c2549 | ||
|
|
330e7f628c | ||
|
|
efe401c7a4 | ||
|
|
b3d88c043d | ||
|
|
edd36dac91 | ||
|
|
6de5fb2885 | ||
|
|
3b2ebafd83 | ||
|
|
dd4f508b77 | ||
|
|
3058a825b7 | ||
|
|
bdbeef3b9a | ||
|
|
8ead4a3c34 | ||
|
|
2d53a9b0dc | ||
|
|
8f47b46219 | ||
|
|
4f0d35bae0 | ||
|
|
de6d907e27 | ||
|
|
c9fc347de1 | ||
|
|
85e67f92c5 | ||
|
|
034a27ce5a | ||
|
|
056abb5901 | ||
|
|
af28406dbc | ||
|
|
376d6ba72c | ||
|
|
736a73453f | ||
|
|
6ced309c5c | ||
|
|
a30593efd6 | ||
|
|
3fa400bc55 | ||
|
|
90ea16325a | ||
|
|
be84a4332a | ||
|
|
cfeba859ac | ||
|
|
38ed7c9bb1 | ||
|
|
315ee90837 | ||
|
|
278574ce91 | ||
|
|
6c225d5469 | ||
|
|
d6a466ac0b | ||
|
|
58526d4faa | ||
|
|
3c90a82e75 | ||
|
|
06a27deebb | ||
|
|
f607f877c9 | ||
|
|
7273041314 | ||
|
|
997d041a84 | ||
|
|
226233ddce | ||
|
|
6c8edcbea5 | ||
|
|
e0f27bb855 | ||
|
|
4a14003f34 | ||
|
|
216a37348b | ||
|
|
ef7b2a9d4f | ||
|
|
ffb1cf4c7c | ||
|
|
964165eabe | ||
|
|
dca2d74eb6 | ||
|
|
e136ff52ab | ||
|
|
0df4944a7b | ||
|
|
2bb296df3b | ||
|
|
52380a3927 | ||
|
|
732f725a24 | ||
|
|
8887b16299 | ||
|
|
75c8b08d9b | ||
|
|
9741dc214d | ||
|
|
8b4b80c725 | ||
|
|
e3b6cccca1 | ||
|
|
d28ef1859d | ||
|
|
74b59f458d | ||
|
|
4ae04c2a9a | ||
|
|
7bd0789402 | ||
|
|
eb9fb9e834 | ||
|
|
c0ef0a7503 | ||
|
|
61719115e5 | ||
|
|
4a8cbe2ab8 | ||
|
|
c966f44189 | ||
|
|
193ef8b232 | ||
|
|
ad70a60d05 | ||
|
|
9dfd5b5526 | ||
|
|
09ec374eec | ||
|
|
8c1e9eda12 | ||
|
|
0eedd55dfd | ||
|
|
58c86d16f3 | ||
|
|
4a9170e981 | ||
|
|
b5b8ff01d5 | ||
|
|
6c86ffdeb7 | ||
|
|
70a1d92d9f | ||
|
|
0749477eb9 | ||
|
|
fb54341a1f | ||
|
|
db601d333b | ||
|
|
af1c2cccac | ||
|
|
bf1b76dcd2 | ||
|
|
48787ab460 | ||
|
|
7d4bf84304 | ||
|
|
5cf5d3e3a2 | ||
|
|
8ddb229447 | ||
|
|
9fdd96af61 | ||
|
|
7fbdd0607f | ||
|
|
0c0a1d8f12 | ||
|
|
30dd9ed267 | ||
|
|
5f0a1d55e5 | ||
|
|
17bef26708 | ||
|
|
58b5620f97 | ||
|
|
327b62ea45 | ||
|
|
91c6d693df | ||
|
|
aa3df3914b | ||
|
|
ad3a024b69 | ||
|
|
b9ec94264e | ||
|
|
fee9e91c4f | ||
|
|
1dc560d45a | ||
|
|
258e2b80f4 | ||
|
|
dcb889704d | ||
|
|
93096c27f9 | ||
|
|
3c6eba561f | ||
|
|
d2714f3338 | ||
|
|
b917db34c7 | ||
|
|
c45abaaa6b | ||
|
|
a2286cb00a | ||
|
|
481787d45c | ||
|
|
756002eef8 | ||
|
|
6789919fca | ||
|
|
d8c615ec65 | ||
|
|
6a1eb0e508 | ||
|
|
984b0260d5 | ||
|
|
785e59e889 | ||
|
|
671be5b191 | ||
|
|
b4e556e63a | ||
|
|
2128943b57 | ||
|
|
6ddb1094b6 | ||
|
|
b9d93b19d4 | ||
|
|
97a5da232d | ||
|
|
0d77af5366 | ||
|
|
a3435f2d22 | ||
|
|
154dd46b6d | ||
|
|
782952d55d | ||
|
|
a6bd293ae8 | ||
|
|
decd112a84 | ||
|
|
f9c15fa0a5 | ||
|
|
1121f2a1fb | ||
|
|
a7989eb79f | ||
|
|
80c199cefb | ||
|
|
317f92f4c8 | ||
|
|
dd344710d4 | ||
|
|
e4e8683c7b | ||
|
|
b2407352a9 | ||
|
|
e1b5e3e112 | ||
|
|
09793d295f | ||
|
|
e9f8c01e04 | ||
|
|
f69821d7db | ||
|
|
1e03219ca3 | ||
|
|
de8b4438e3 | ||
|
|
2cfd3330f0 | ||
|
|
1988b432ab | ||
|
|
6b2363d3f6 | ||
|
|
e76af56c86 | ||
|
|
b7173a3ae2 | ||
|
|
3f71517e1f | ||
|
|
17fbe52cf6 | ||
|
|
ecc526a9d8 | ||
|
|
436f4aa953 | ||
|
|
fd10ce0600 | ||
|
|
d9cd40eb48 | ||
|
|
9b9ab16b55 | ||
|
|
14256e7a90 | ||
|
|
d36c387c1f | ||
|
|
7e61b172b2 | ||
|
|
bd0cae9298 | ||
|
|
6ae94581bb | ||
|
|
ea8ae2bf2c | ||
|
|
694bc3312e | ||
|
|
f565939437 | ||
|
|
272d2747cd | ||
|
|
675956c92d | ||
|
|
03410ed4a2 | ||
|
|
ac88e7fabf | ||
|
|
71d42d7e73 | ||
|
|
2613762ac2 | ||
|
|
831b21cd39 | ||
|
|
7869d80073 | ||
|
|
ea39d80e83 | ||
|
|
62e481a608 | ||
|
|
bac7148d69 | ||
|
|
62f16a3c4b | ||
|
|
248ecd54e4 | ||
|
|
1f59accb4e | ||
|
|
dd7e4375cf | ||
|
|
3df3999138 | ||
|
|
1ca6443c9d | ||
|
|
a804cb5d07 | ||
|
|
a0883c9615 | ||
|
|
03c968a8fe | ||
|
|
e34de22cf4 | ||
|
|
3db1f1e01f | ||
|
|
5e467c737f | ||
|
|
dfc42fa7c5 | ||
|
|
f1016e82c7 | ||
|
|
5e099a5dbd | ||
|
|
8dceec9316 | ||
|
|
6f51092998 | ||
|
|
4e83ca67f5 | ||
|
|
76e6c6f8ea | ||
|
|
343a529e63 | ||
|
|
c6ccc68607 | ||
|
|
8009f4a6ee | ||
|
|
e126cecb16 | ||
|
|
5e19202540 | ||
|
|
06466bac0e | ||
|
|
156edcacb2 | ||
|
|
bf4b8d246a | ||
|
|
abc37ec35e | ||
|
|
dd735464a2 | ||
|
|
7041bdb144 | ||
|
|
19e8c3fcb6 | ||
|
|
06c5135d4e | ||
|
|
f66066cbd8 | ||
|
|
a144622a06 | ||
|
|
05c814b8c1 | ||
|
|
e0af51fb06 | ||
|
|
7b749558df | ||
|
|
1797c3c617 | ||
|
|
28a1c28cb4 | ||
|
|
20fcbb9c62 | ||
|
|
7c9c97f2b9 | ||
|
|
d72e121fd4 | ||
|
|
003fc3dab5 | ||
|
|
2c883d7cdc | ||
|
|
fcba49768c | ||
|
|
959af9c3af | ||
|
|
1edacf05e9 | ||
|
|
1f756fb386 | ||
|
|
5ebdc01783 | ||
|
|
aa29201e00 | ||
|
|
b98acdf97a | ||
|
|
7aeb4788ed | ||
|
|
7b1db40d4a | ||
|
|
dcaa90a855 | ||
|
|
0a74efe3f8 | ||
|
|
5a73134c7c | ||
|
|
43d535cf8d | ||
|
|
4a71edf7bc | ||
|
|
ae90040ed2 | ||
|
|
fbf982f001 | ||
|
|
839e3ac775 | ||
|
|
65d230f32b | ||
|
|
0882b0db0c | ||
|
|
3d92c1fb32 | ||
|
|
d4e679432c | ||
|
|
54c8cb9fd7 | ||
|
|
221ee1a122 | ||
|
|
2b29f90c92 | ||
|
|
d7ec0e570f | ||
|
|
86c2144943 | ||
|
|
ee85230db2 | ||
|
|
fde99a8dc1 | ||
|
|
2e692a1146 | ||
|
|
73f36fb5b3 | ||
|
|
f866ad52dc | ||
|
|
9566910574 | ||
|
|
4188d4d10b | ||
|
|
d75152cb74 | ||
|
|
add54b8089 | ||
|
|
7431495f13 | ||
|
|
5ed5566106 | ||
|
|
2d9a4412a4 | ||
|
|
982abd5719 | ||
|
|
f09db0cbe5 | ||
|
|
80cacc462e | ||
|
|
1a2b2f8870 | ||
|
|
e8c576047f | ||
|
|
995d3152eb | ||
|
|
ddc40cc273 | ||
|
|
5ee311b831 | ||
|
|
ea90296e22 | ||
|
|
58c1a8d148 | ||
|
|
87a19c7938 | ||
|
|
5b0709de2f | ||
|
|
87db37c65e | ||
|
|
100db13f99 | ||
|
|
d2061b27a2 | ||
|
|
002dfb9b84 | ||
|
|
0ab1241f09 | ||
|
|
aa63656ff1 | ||
|
|
6ba32da73a | ||
|
|
1129e71069 | ||
|
|
3b32fd5c49 | ||
|
|
7df1ed271f | ||
|
|
67e9b40bab | ||
|
|
62b80903f0 | ||
|
|
993d917478 | ||
|
|
ac20880a14 | ||
|
|
99cfe05ee5 | ||
|
|
7c187e82b7 | ||
|
|
2a760660a9 | ||
|
|
4b11826a7d | ||
|
|
887f586874 | ||
|
|
db94c04179 | ||
|
|
51e64c69f3 | ||
|
|
97c46843d2 | ||
|
|
16310c91c9 | ||
|
|
220e421c22 | ||
|
|
a156752fae | ||
|
|
5d1c574dc5 | ||
|
|
6310c20217 | ||
|
|
0dfc30fdd8 | ||
|
|
f050571696 | ||
|
|
6db5bc3721 | ||
|
|
4f84c281ef | ||
|
|
0ca438e081 | ||
|
|
2472aee473 | ||
|
|
714856f74c | ||
|
|
e98932df92 | ||
|
|
95623dac63 | ||
|
|
827015a8e3 | ||
|
|
a379818386 | ||
|
|
4516e4e9f6 | ||
|
|
83e44961fc | ||
|
|
bc2e959897 | ||
|
|
1b0b4ba416 | ||
|
|
25a0752e58 | ||
|
|
b0a7e0bbcc | ||
|
|
af561abc71 | ||
|
|
3bf323aa2b | ||
|
|
c9a33c638a | ||
|
|
8d1042859d | ||
|
|
9072571fc1 | ||
|
|
d8748b8216 | ||
|
|
2f930b201a | ||
|
|
26b4195f1f | ||
|
|
5c078d13c0 | ||
|
|
2825ac282a | ||
|
|
cd31a79f41 | ||
|
|
52eb9f46d2 | ||
|
|
08c430dcfb | ||
|
|
b486aeeac5 | ||
|
|
13da6102b3 | ||
|
|
1e1c4e017e | ||
|
|
ceaaab5a86 | ||
|
|
2b7d03d4f9 | ||
|
|
c8810557c1 | ||
|
|
2a4bfe49f5 | ||
|
|
b71879e90e | ||
|
|
63ae262752 | ||
|
|
2c3a4fef26 | ||
|
|
e696e32b95 | ||
|
|
3f9df49e3e | ||
|
|
1ac3d3c5a6 | ||
|
|
b3d1aade71 | ||
|
|
c300c723f7 | ||
|
|
d0ec069dcb | ||
|
|
6811406b24 | ||
|
|
0879994aff | ||
|
|
b2b5ccf69c | ||
|
|
6f5d4fbf30 | ||
|
|
8ad1ea499d | ||
|
|
4df17abea7 | ||
|
|
972c8a97dc | ||
|
|
135d17fbd3 | ||
|
|
095b2a8926 | ||
|
|
297677c2c6 | ||
|
|
2a00f23459 | ||
|
|
68ad916a91 | ||
|
|
0f3883152b | ||
|
|
60baee4020 | ||
|
|
5a7ff4780f | ||
|
|
dd2845401c | ||
|
|
a47c947065 | ||
|
|
203ac51681 | ||
|
|
a7986993bf | ||
|
|
ce59d40570 | ||
|
|
19f33e22d7 | ||
|
|
7d818e18be | ||
|
|
52af0aceea | ||
|
|
cd1cf6f615 | ||
|
|
2f925aeb22 | ||
|
|
f99b42da5e | ||
|
|
40beef061f | ||
|
|
f0a434c167 | ||
|
|
92b66dbf17 | ||
|
|
f336261d6b | ||
|
|
c6324b83ea | ||
|
|
54873992ab | ||
|
|
38d568a807 | ||
|
|
dcef1212c6 | ||
|
|
2030b70ff8 | ||
|
|
41a188eaec | ||
|
|
f3ddaf455c | ||
|
|
aa979e4bc7 | ||
|
|
baddc56f01 | ||
|
|
9836e43d10 | ||
|
|
c8cecd9f46 | ||
|
|
38c59a44de | ||
|
|
fddc86c2b1 | ||
|
|
0dddc9b5cb | ||
|
|
8b14bd4e87 | ||
|
|
5c77969e83 | ||
|
|
e049596252 | ||
|
|
afa1327242 | ||
|
|
da3b7f5a41 | ||
|
|
100f61ee24 | ||
|
|
dcd2794ff5 | ||
|
|
8d1d3fe12b | ||
|
|
87ae0f058a | ||
|
|
7e1be6625f | ||
|
|
0c6fcd1678 | ||
|
|
37ac46273d | ||
|
|
1e21416ccb | ||
|
|
e2353959c1 | ||
|
|
61d3d5b068 | ||
|
|
572e5ed9c6 | ||
|
|
4dbfa2febb | ||
|
|
0293d2027d | ||
|
|
c28b94890c | ||
|
|
63a910d2fc | ||
|
|
fbef5c6a6a | ||
|
|
674b6e9f43 | ||
|
|
7897b6ad55 | ||
|
|
6769b54e1c | ||
|
|
734ab4429f | ||
|
|
5db6a64f3f | ||
|
|
dcc458ea94 | ||
|
|
7354e7e3c9 | ||
|
|
07ef765f7e | ||
|
|
c58ad8b593 | ||
|
|
31091c1053 | ||
|
|
c95f9f5e6a | ||
|
|
93c3eb064d | ||
|
|
b13162c5bf | ||
|
|
ee1725684a | ||
|
|
a07b234659 |
No files matched your search
@@ -7,3 +7,6 @@ FEXCore/Source/Interface/Core/X86Tables/*
|
||||
# Inline headers with list-like content that can't be processed individually
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
|
||||
|
||||
# Include files in unittests
|
||||
unittests/*ASM/Includes/*.inc
|
||||
@@ -20,3 +20,5 @@
|
||||
# Whole-tree reformat with clang-format-19
|
||||
5267cde60e7642852d18f20ae8568643bb5293d5
|
||||
|
||||
# Minor reformat with clang-format-19
|
||||
9fdd96af61c969cb5732471223f00eda64b7a069
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
build_plus_test:
|
||||
@@ -33,7 +34,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
@@ -136,6 +136,9 @@ jobs:
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
# These tests require non-portable install due to thunks.
|
||||
FEX_PORTABLE: 0
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
|
||||
@@ -20,6 +20,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
glibc_fault_test:
|
||||
@@ -40,7 +41,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
hostrunner_tests:
|
||||
@@ -33,7 +34,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -33,7 +33,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -48,7 +48,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
@@ -78,7 +77,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -60,7 +60,6 @@ jobs:
|
||||
START_REV: ${{ github.event.pull_request.base.sha }}
|
||||
END_REV: ${{ github.event.pull_request.head.sha }}
|
||||
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
|
||||
# Using --diff_from_common_commit option available in clang-format-19
|
||||
run: |
|
||||
python ./External/code-format-helper/code-format-helper.py \
|
||||
--repo "FEX-emu/FEX" \
|
||||
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
vixl_simulator:
|
||||
@@ -34,7 +35,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -46,12 +46,12 @@ jobs:
|
||||
- name: Configure CMake arm64ec
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Configure CMake wow64
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_wow64
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build arm64ec
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
|
||||
@@ -46,3 +46,6 @@
|
||||
[submodule "External/tracy"]
|
||||
path = External/tracy
|
||||
url = https://github.com/wolfpld/tracy
|
||||
[submodule "External/range-v3"]
|
||||
path = External/range-v3
|
||||
url = https://github.com/ericniebler/range-v3.git
|
||||
+19
-55
@@ -4,7 +4,6 @@ project(FEX C CXX ASM)
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
@@ -304,7 +303,8 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
include(CTest)
|
||||
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
@@ -319,7 +319,7 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
|
||||
@@ -335,7 +335,7 @@ endif()
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
find_package(Catch2 3 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
@@ -345,6 +345,9 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
include(Catch)
|
||||
else ()
|
||||
# Override any previously generated test list to avoid running stale test binaries
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
find_package(fmt QUIET)
|
||||
@@ -354,6 +357,12 @@ if (NOT fmt_FOUND)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
|
||||
find_package(range-v3 QUIET)
|
||||
if (NOT range-v3_FOUND)
|
||||
add_subdirectory(External/range-v3/)
|
||||
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
include_directories(External/tiny-json/)
|
||||
|
||||
@@ -449,13 +458,8 @@ endif()
|
||||
|
||||
add_compile_options(-Wall)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
message(STATUS "Unit tests are enabled")
|
||||
if (NOT BUILD_TESTING)
|
||||
# CMake checks this variable before generating CTestTestfile.cmake
|
||||
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
|
||||
endif()
|
||||
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
@@ -486,10 +490,11 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
DESTINATION ${DATA_DIRECTORY}/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
@@ -550,6 +555,7 @@ if (BUILD_THUNKS)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -559,6 +565,7 @@ if (BUILD_THUNKS)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
@@ -600,46 +607,3 @@ if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Parse the version here
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
|
||||
endif()
|
||||
include (CPack)
|
||||
@@ -36,24 +36,33 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
if (IsADRRange(Imm)) {
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adr(rd, &Label->Backward);
|
||||
return adr(rd, &Label->Backward);
|
||||
} else {
|
||||
adr(rd, &Label->Forward);
|
||||
return adr(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -62,32 +71,42 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) {
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adrp(rd, &Label->Backward);
|
||||
return adrp(rd, &Label->Backward);
|
||||
} else {
|
||||
adrp(rd, &Label->Forward);
|
||||
return adrp(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
adr(rd, Label);
|
||||
return adr(rd, Label);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
@@ -102,23 +121,28 @@ public:
|
||||
// Now even an add
|
||||
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset too large");
|
||||
FEX_UNREACHABLE;
|
||||
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
LongAddressGen(rd, &Label->Backward);
|
||||
return LongAddressGen(rd, &Label->Backward);
|
||||
} else {
|
||||
LongAddressGen(rd, &Label->Forward);
|
||||
return LongAddressGen(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -174,7 +198,7 @@ public:
|
||||
// Logical immediate
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
and_(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -185,7 +209,7 @@ public:
|
||||
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
ands(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -196,14 +220,14 @@ public:
|
||||
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
orr(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -333,7 +357,7 @@ public:
|
||||
bfi(s, rd, Reg::zr, lsb, width);
|
||||
}
|
||||
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
@@ -862,12 +886,6 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
@@ -977,7 +995,7 @@ private:
|
||||
}
|
||||
|
||||
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits, lsb, width);
|
||||
|
||||
@@ -2244,8 +2244,7 @@ public:
|
||||
|
||||
template<IsQOrDRegister T>
|
||||
void movi(SubRegSize size, T rd, uint64_t Imm, uint16_t Shift = 0) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit ||
|
||||
size == SubRegSize::i64Bit,
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Unsupported movi size");
|
||||
|
||||
uint32_t cmode;
|
||||
|
||||
@@ -20,23 +20,31 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
return b(Cond, &Label->Backward);
|
||||
} else {
|
||||
b(Cond, &Label->Forward);
|
||||
return b(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,24 +53,32 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
return bc(Cond, &Label->Backward);
|
||||
} else {
|
||||
bc(Cond, &Label->Forward);
|
||||
return bc(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -98,25 +114,32 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void b(ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
return b(&Label->Backward);
|
||||
} else {
|
||||
b(&Label->Forward);
|
||||
return b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -126,25 +149,33 @@ public:
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void bl(ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
return bl(&Label->Backward);
|
||||
} else {
|
||||
bl(&Label->Forward);
|
||||
return bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -155,28 +186,35 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
return cbz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
return cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -186,28 +224,35 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
return cbnz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
return cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -217,28 +262,35 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
return tbz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
return tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -247,27 +299,35 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
return tbnz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
return tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <type_traits>
|
||||
@@ -585,6 +586,15 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
|
||||
template<typename T>
|
||||
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
|
||||
|
||||
template<typename T>
|
||||
concept IsLabel = std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>;
|
||||
|
||||
enum class BranchEncodeSucceeded {
|
||||
Success,
|
||||
Failure,
|
||||
};
|
||||
|
||||
// Whether or not a given set of vector registers are sequential
|
||||
// in increasing order as far as the register file is concerned (modulo its size)
|
||||
//
|
||||
@@ -637,19 +647,25 @@ public:
|
||||
|
||||
// Bind a backward label to an address.
|
||||
// Address that is bound is the current emitter location.
|
||||
void Bind(BackwardLabel* Label) {
|
||||
[[nodiscard]] bool Bind(BackwardLabel* Label) {
|
||||
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Always binds because it is only storing a location.
|
||||
return true;
|
||||
}
|
||||
|
||||
void Bind(const ForwardLabel::Reference* Label) {
|
||||
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
|
||||
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
// Patch up the instructions
|
||||
switch (Label->Type) {
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
if (!IsADRRange(Imm)) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
@@ -661,7 +677,12 @@ public:
|
||||
case ForwardLabel::InstType::ADRP: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
@@ -671,11 +692,13 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FF'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -685,11 +708,13 @@ public:
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -703,7 +728,10 @@ public:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x7'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -752,27 +780,41 @@ public:
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
void Bind(ForwardLabel* Label) {
|
||||
[[nodiscard]] bool Bind(ForwardLabel* Label) {
|
||||
bool Bound = true;
|
||||
if (Label->FirstInst.Location) {
|
||||
Bind(&Label->FirstInst);
|
||||
Bound &= Bind(&Label->FirstInst);
|
||||
}
|
||||
for (auto& Inst : Label->Insts) {
|
||||
Bind(&Inst);
|
||||
Bound &= Bind(&Inst);
|
||||
}
|
||||
|
||||
return Bound;
|
||||
}
|
||||
|
||||
// Bind a bidirectional location to a location.
|
||||
// Binds both forwards and backwards depending on how the label was used.
|
||||
void Bind(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
|
||||
bool Bound = true;
|
||||
if (!Label->Backward.Location) {
|
||||
Bind(&Label->Backward);
|
||||
Bound &= Bind(&Label->Backward);
|
||||
}
|
||||
Bind(&Label->Forward);
|
||||
Bound &= Bind(&Label->Forward);
|
||||
|
||||
return Bound;
|
||||
}
|
||||
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
#include <CodeEmitter/VixlUtils.inl>
|
||||
|
||||
@@ -1541,7 +1541,7 @@ public:
|
||||
void sqincp(SubRegSize size, XRegister rdn, PRegister pm) {
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b00, size, rdn, pm);
|
||||
}
|
||||
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
|
||||
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
|
||||
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b00, size, rdn, pm);
|
||||
}
|
||||
@@ -1554,7 +1554,7 @@ public:
|
||||
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm) {
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b10, size, rdn, pm);
|
||||
}
|
||||
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
|
||||
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
|
||||
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b10, size, rdn, pm);
|
||||
}
|
||||
@@ -3296,7 +3296,7 @@ private:
|
||||
const auto log2_size_bytes = FEXCore::ilog2(size_bytes);
|
||||
|
||||
// We can index up to 512-bit registers with dup
|
||||
[[maybe_unused]] const auto max_index = (64U >> log2_size_bytes) - 1;
|
||||
const auto max_index = (64U >> log2_size_bytes) - 1;
|
||||
LOGMAN_THROW_A_FMT(Index <= max_index, "dup index ({}) too large. Must be within [0, {}].", Index, max_index);
|
||||
|
||||
// imm2:tsz make up a 7 bit wide field, with each increasing element size
|
||||
@@ -3326,7 +3326,7 @@ private:
|
||||
|
||||
uint32_t shift = 0;
|
||||
if (!is_uint8_imm) {
|
||||
[[maybe_unused]] const bool is_uint16_imm = (imm >> 16) == 0;
|
||||
const bool is_uint16_imm = (imm >> 16) == 0;
|
||||
|
||||
LOGMAN_THROW_A_FMT(is_uint16_imm, "Immediate ({}) must be a 16-bit value within [256, 65280]", imm);
|
||||
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
|
||||
@@ -4152,7 +4152,7 @@ private:
|
||||
|
||||
const auto& op_data = mem_op.MetaType.ScalarVectorType;
|
||||
const bool is_scaled = op_data.scale != 0;
|
||||
[[maybe_unused]] const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
|
||||
LOGMAN_THROW_A_FMT(op_data.scale == 0 || op_data.scale == msize_value, "scale may only be 0 or {}", msize_value);
|
||||
|
||||
@@ -4266,7 +4266,7 @@ private:
|
||||
const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
const auto msize_bytes = 1U << msize_value;
|
||||
|
||||
[[maybe_unused]] const auto imm_limit = (32U << msize_value) - msize_bytes;
|
||||
const auto imm_limit = (32U << msize_value) - msize_bytes;
|
||||
const auto imm = mem_op.MetaType.VectorImmType.Imm;
|
||||
const auto imm_to_encode = imm >> msize_value;
|
||||
|
||||
@@ -4332,8 +4332,8 @@ private:
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_A_FMT((imm % num_regs) == 0, "Offset must be a multiple of {}", num_regs);
|
||||
|
||||
[[maybe_unused]] const auto min_offset = -8 * num_regs;
|
||||
[[maybe_unused]] const auto max_offset = 7 * num_regs;
|
||||
const auto min_offset = -8 * num_regs;
|
||||
const auto max_offset = 7 * num_regs;
|
||||
LOGMAN_THROW_A_FMT(imm >= min_offset && imm <= max_offset,
|
||||
"Invalid load/store offset ({}). Offset must be a multiple of {} and be within [{}, {}]", imm, num_regs, min_offset,
|
||||
max_offset);
|
||||
@@ -4440,8 +4440,8 @@ private:
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
|
||||
const auto esize = static_cast<int>(16 << ssz);
|
||||
[[maybe_unused]] const auto max_imm = (esize << 3) - esize;
|
||||
[[maybe_unused]] const auto min_imm = -(max_imm + esize);
|
||||
const auto max_imm = (esize << 3) - esize;
|
||||
const auto min_imm = -(max_imm + esize);
|
||||
|
||||
LOGMAN_THROW_A_FMT((imm % esize) == 0, "imm ({}) must be a multiple of {}", imm, esize);
|
||||
LOGMAN_THROW_A_FMT(imm >= min_imm && imm <= max_imm, "imm ({}) must be within [{}, {}]", imm, min_imm, max_imm);
|
||||
@@ -4485,7 +4485,7 @@ private:
|
||||
const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
|
||||
const auto data_size_bytes = 1U << msize_value;
|
||||
[[maybe_unused]] const auto max_imm = (64U << msize_value) - data_size_bytes;
|
||||
const auto max_imm = (64U << msize_value) - data_size_bytes;
|
||||
LOGMAN_THROW_A_FMT((imm % data_size_bytes) == 0 && imm <= max_imm, "imm must be a multiple of {} and be within [0, {}]",
|
||||
data_size_bytes, max_imm);
|
||||
|
||||
@@ -4861,7 +4861,7 @@ private:
|
||||
"64-bit variants may only use Zm between z0-z15");
|
||||
|
||||
const auto Underlying = FEXCore::ToUnderlying(size);
|
||||
[[maybe_unused]] const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
|
||||
const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
|
||||
LOGMAN_THROW_A_FMT(index <= IndexMax, "Index must be within 0-{}", IndexMax);
|
||||
|
||||
// Can be bit 20 or 19 depending on whether or not the element size is 64-bit.
|
||||
@@ -5117,14 +5117,15 @@ private:
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>;
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Determines if a floating-point value is capable of being converted
|
||||
// into an 8-bit immediate. See pseudocode definition of VFPExpandImm
|
||||
// in ARM A-profile reference manual for a general overview of how this was derived.
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
[[nodiscard, maybe_unused]]
|
||||
[[nodiscard]]
|
||||
static bool IsValidFPValueForImm8(T value) {
|
||||
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
|
||||
|
||||
static constexpr std::array mantissa_masks {
|
||||
@@ -5162,12 +5163,15 @@ private:
|
||||
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
protected:
|
||||
static uint32_t FP32ToImm8(float value) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint32_t>(value);
|
||||
const auto bits = std::bit_cast<uint32_t>(value);
|
||||
const auto sign = (bits & 0x80000000) >> 24;
|
||||
const auto expb2 = (bits & 0x20000000) >> 23;
|
||||
const auto b5_to_0 = (bits >> 19) & 0x3F;
|
||||
@@ -5176,9 +5180,11 @@ protected:
|
||||
}
|
||||
|
||||
static uint32_t FP64ToImm8(double value) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint64_t>(value);
|
||||
const auto bits = std::bit_cast<uint64_t>(value);
|
||||
const auto sign = (bits & 0x80000000'00000000) >> 56;
|
||||
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
|
||||
const auto b5_to_0 = (bits >> 48) & 0x3F;
|
||||
@@ -5202,7 +5208,7 @@ private:
|
||||
uint32_t shift = 0;
|
||||
if (!is_int8_imm) {
|
||||
const int32_t imm16_limit = 32768;
|
||||
[[maybe_unused]] const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
|
||||
const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(is_int16_imm, "Immediate ({}) must be a 16-bit value within [-32768, 32512]", imm);
|
||||
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
|
||||
|
||||
@@ -30,7 +30,7 @@ public:
|
||||
const uint32_t SizeImm = FEXCore::ToUnderlying(size);
|
||||
const uint32_t IndexShift = SizeImm + 1;
|
||||
const uint32_t ElementSize = 1U << SizeImm;
|
||||
[[maybe_unused]] const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
|
||||
|
||||
@@ -1381,7 +1381,7 @@ private:
|
||||
void ASIMDScalarXIndexedElement(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rm, VRegister rn, VRegister rd, uint32_t index) {
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i8Bit, "Scalar size must not be 8-bit");
|
||||
|
||||
[[maybe_unused]] const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
|
||||
const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
|
||||
LOGMAN_THROW_A_FMT(index < invalid_bound, "Index ({}) must be within [0-{}]", index, invalid_bound - 1);
|
||||
|
||||
uint32_t Instr = 0b0101'1111'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
@@ -41,7 +41,6 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
[[maybe_unused]] constexpr auto kDRegSize = 64;
|
||||
|
||||
constexpr auto kWRegSize = 32;
|
||||
constexpr auto kXRegSize = 64;
|
||||
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) || (width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
|
||||
@@ -129,8 +128,8 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
// Compute the repeat distance d, and set up a bitmask covering the basic
|
||||
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
|
||||
// of these cases the N bit of the output will be zero.
|
||||
clz_a = CountLeadingZeros(a, kXRegSize);
|
||||
int clz_c = CountLeadingZeros(c, kXRegSize);
|
||||
clz_a = std::countl_zero(a);
|
||||
int clz_c = std::countl_zero(c);
|
||||
d = clz_a - clz_c;
|
||||
mask = ((UINT64_C(1) << d) - 1);
|
||||
out_n = 0;
|
||||
@@ -151,7 +150,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
// of set bits in our word, meaning that we have the trivial case of
|
||||
// d == 64 and only one 'repetition'. Set up all the same variables as in
|
||||
// the general case above, and set the N bit in the output.
|
||||
clz_a = CountLeadingZeros(a, kXRegSize);
|
||||
clz_a = std::countl_zero(a);
|
||||
d = 64;
|
||||
mask = ~UINT64_C(0);
|
||||
out_n = 1;
|
||||
@@ -159,7 +158,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
}
|
||||
|
||||
// If the repeat period d is not a power of two, it can't be encoded.
|
||||
if (!IsPowerOf2(d)) {
|
||||
if (!std::has_single_bit(uint32_t(d))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -179,7 +178,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
static const uint64_t multipliers[] = {
|
||||
0x0000000000000001UL, 0x0000000100000001UL, 0x0001000100010001UL, 0x0101010101010101UL, 0x1111111111111111UL, 0x5555555555555555UL,
|
||||
};
|
||||
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
|
||||
uint64_t multiplier = multipliers[std::countl_zero(uint64_t(d)) - 57];
|
||||
uint64_t candidate = (b - a) * multiplier;
|
||||
|
||||
if (value != candidate) {
|
||||
@@ -194,7 +193,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
// Count the set bits in our basic stretch. The special case of clz(0) == -1
|
||||
// makes the answer come out right for stretches that reach the very top of
|
||||
// the word (e.g. numbers like 0xffffc00000000000).
|
||||
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
|
||||
int clz_b = (b == 0) ? -1 : std::countl_zero(b);
|
||||
int s = clz_a - clz_b;
|
||||
|
||||
// Decide how many bits to rotate right by, to put the low bit of that basic
|
||||
@@ -285,11 +284,6 @@ INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
|
||||
private:
|
||||
|
||||
template<typename V>
|
||||
static inline bool IsPowerOf2(V value) {
|
||||
return (value != 0) && ((value & (value - 1)) == 0);
|
||||
}
|
||||
|
||||
// Some compilers dislike negating unsigned integers,
|
||||
// so we provide an equivalent.
|
||||
template<typename T>
|
||||
@@ -302,50 +296,4 @@ static inline uint64_t LowestSetBit(uint64_t value) {
|
||||
return value & UnsignedNegate(value);
|
||||
}
|
||||
|
||||
template<typename V>
|
||||
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
|
||||
#if COMPILER_HAS_BUILTIN_CLZ
|
||||
if (width == 32) {
|
||||
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
|
||||
} else if (width == 64) {
|
||||
return (value == 0) ? 64 : __builtin_clzll(value);
|
||||
}
|
||||
#endif
|
||||
return CountLeadingZerosFallBack(value, width);
|
||||
}
|
||||
|
||||
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
|
||||
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
|
||||
if (value == 0) {
|
||||
return width;
|
||||
}
|
||||
int count = 0;
|
||||
value = value << (64 - width);
|
||||
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
|
||||
count += 32;
|
||||
value = value << 32;
|
||||
}
|
||||
if ((value & UINT64_C(0xffff000000000000)) == 0) {
|
||||
count += 16;
|
||||
value = value << 16;
|
||||
}
|
||||
if ((value & UINT64_C(0xff00000000000000)) == 0) {
|
||||
count += 8;
|
||||
value = value << 8;
|
||||
}
|
||||
if ((value & UINT64_C(0xf000000000000000)) == 0) {
|
||||
count += 4;
|
||||
value = value << 4;
|
||||
}
|
||||
if ((value & UINT64_C(0xc000000000000000)) == 0) {
|
||||
count += 2;
|
||||
value = value << 2;
|
||||
}
|
||||
if ((value & UINT64_C(0x8000000000000000)) == 0) {
|
||||
count += 1;
|
||||
}
|
||||
count += (value == 0);
|
||||
return count;
|
||||
}
|
||||
|
||||
public:
|
||||
@@ -4,7 +4,8 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
|
||||
# Any configuration file json file that needs to be generated
|
||||
@@ -21,5 +22,6 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
@@ -1,3 +0,0 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
|
||||
@@ -1,18 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Setup binfmt_misc
|
||||
update-binfmts --import FEX-x86
|
||||
update-binfmts --import FEX-x86_64
|
||||
}
|
||||
|
||||
# Install FEXInterpreter hardlink
|
||||
# Needs to be done before setting up binfmt_misc
|
||||
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
@@ -1,17 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Uninstall
|
||||
update-binfmts --unimport FEX-x86
|
||||
update-binfmts --unimport FEX-x86_64
|
||||
}
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
|
||||
# Remove FEXInterpreter hardlink
|
||||
unlink /usr/bin/FEXInterpreter
|
||||
@@ -1 +0,0 @@
|
||||
activate-noawait ldconfig
|
||||
+1
-1
@@ -14,7 +14,7 @@ RUN mkdir build
|
||||
|
||||
ARG CC=clang-13
|
||||
ARG CXX=clang++-13
|
||||
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
|
||||
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
|
||||
RUN ninja
|
||||
|
||||
WORKDIR /FEX/build
|
||||
|
||||
@@ -10,7 +10,8 @@ function(GenBinFmt Name)
|
||||
# Then install the configured binfmt
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
|
||||
COMPONENT Runtime)
|
||||
endfunction()
|
||||
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
@@ -19,7 +20,8 @@ if (NOT USE_LEGACY_BINFMTMISC)
|
||||
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
|
||||
COMPONENT Runtime)
|
||||
else()
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
|
||||
@@ -1 +1 @@
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
|
||||
@@ -1,5 +1,5 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
|
||||
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
|
||||
@@ -1 +1 @@
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
|
||||
@@ -1,5 +1,5 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
|
||||
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
|
||||
let
|
||||
toolchain = pkgs.fetchzip {
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
|
||||
};
|
||||
|
||||
cmakeToolchainFile = pkgs.substitute {
|
||||
@@ -45,7 +45,7 @@ pkgs.mkShell {
|
||||
fi
|
||||
'';
|
||||
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
|
||||
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
|
||||
|
||||
@@ -18,4 +18,4 @@ then
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
|
||||
@@ -18,4 +18,4 @@ then
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
|
||||
@@ -14,4 +14,4 @@ fi
|
||||
rm -rf unittests/FEXLinuxTests
|
||||
|
||||
set -o xtrace
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
@@ -214,7 +214,6 @@ class ClangFormatHelper(FormatHelper):
|
||||
self.clang_fmt_path,
|
||||
"--binary=clang-format-19",
|
||||
"--diff",
|
||||
"--diff_from_common_commit",
|
||||
]
|
||||
|
||||
if args.start_rev and args.end_rev:
|
||||
|
||||
Vendored
+1
-1
Submodule External/drm-headers updated: 0675d2f291...3e49836995.
Vendored
+1
-1
Submodule External/fmt updated: 20c8fdad06...e424e3f2e6.
+1
Submodule External/range-v3 added at ca1388fb9d.
Vendored
+1
-1
Submodule External/vixl updated: 84bc10c107...ed690c9eca.
@@ -78,6 +78,6 @@ install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
@@ -118,41 +118,6 @@ def print_man_env_option(name, desc, default, no_json_key):
|
||||
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
|
||||
output_man.write(".Pp\n\n")
|
||||
|
||||
def print_man_options(options):
|
||||
output_man.write(".Sh OPTIONS\n")
|
||||
output_man.write(".Bl -tag -width -indent\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
short = None
|
||||
long = op_key.lower()
|
||||
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_option(
|
||||
short,
|
||||
long,
|
||||
op_vals["Desc"],
|
||||
default
|
||||
)
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
|
||||
output_man.write("\n.sp\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
def print_man_environment(options):
|
||||
output_man.write(".Sh ENVIRONMENT\n")
|
||||
output_man.write(".Bl -tag -width -indent\n")
|
||||
@@ -194,7 +159,7 @@ def print_man_environment_tail():
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
@@ -208,7 +173,7 @@ def print_man_environment_tail():
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
@@ -227,8 +192,8 @@ def print_man_environment_tail():
|
||||
"PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For FEX on Linux:",
|
||||
"These files are instead read from <FEXPath>/fex-emu/ by default.",
|
||||
"For Arm64ec/Wow64 WINE builds:",
|
||||
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
@@ -240,20 +205,12 @@ def print_man_header():
|
||||
.Dt FEX
|
||||
.Os Linux
|
||||
.Sh NAME
|
||||
.Nm FEXLoader
|
||||
.Nm FEXInterpreter
|
||||
.Nm FEX
|
||||
.Nm FEXBash
|
||||
.Nd Fast x86-64 and x86 emulation.
|
||||
.Sh SYNOPSIS
|
||||
.Nm
|
||||
.Op options
|
||||
.Op Ar --
|
||||
.Ar Application
|
||||
<args> ...
|
||||
.Pp
|
||||
.Nm FEXInterpreter
|
||||
.Ar Application
|
||||
<args> ...
|
||||
.Ar <args> ...
|
||||
.Pp
|
||||
.Nm FEXBash
|
||||
.Ar <args> ...
|
||||
@@ -361,82 +318,6 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
|
||||
|
||||
output_argloader.write("\n");
|
||||
|
||||
def print_argloader_options(options):
|
||||
output_argloader.write("#ifdef BEFORE_PARSE\n")
|
||||
output_argloader.write("#undef BEFORE_PARSE\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "\"" + default + "\""
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = "\"" + op_vals["TextDefault"] + "\""
|
||||
|
||||
short = None
|
||||
choices = None
|
||||
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
if ("Choices" in op_vals):
|
||||
choices = op_vals["Choices"]
|
||||
|
||||
print_config_option(
|
||||
op_vals["Type"],
|
||||
op_group,
|
||||
op_key,
|
||||
default,
|
||||
short,
|
||||
choices,
|
||||
op_vals["Desc"])
|
||||
|
||||
output_argloader.write("\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_argloader_options(options):
|
||||
output_argloader.write("#ifdef AFTER_PARSE\n")
|
||||
output_argloader.write("#undef AFTER_PARSE\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
|
||||
|
||||
value_type = op_vals["Type"]
|
||||
NeedsString = False
|
||||
conversion_func = "fextl::fmt::format(\"{}\", "
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "std::move("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key, op_key))
|
||||
elif (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
else:
|
||||
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
|
||||
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
|
||||
def print_parse_envloader_options(options):
|
||||
output_argloader.write("#ifdef ENVLOADER\n")
|
||||
output_argloader.write("#undef ENVLOADER\n")
|
||||
@@ -447,13 +328,13 @@ def print_parse_envloader_options(options):
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tValue = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = {0}(Value_View);\n".format(conversion_func))
|
||||
output_argloader.write("\tValue = {0}(Value_View);\n".format(conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
@@ -467,15 +348,15 @@ def print_parse_jsonloader_options(options):
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
elif (value_type == "strarray"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
assert op_key is not None, "No options found in JSONLOADER"
|
||||
output_argloader.write("else {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("else {\n")
|
||||
output_argloader.write("\tSet(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
@@ -517,41 +398,6 @@ def print_parse_enum_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
|
||||
# Spin through all the items and see if we have a duplicate option
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
short = None
|
||||
long = op_key.lower()
|
||||
long_invert = None
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
if (op_vals["Type"] == "bool"):
|
||||
long_invert = "no-" + long
|
||||
|
||||
# Check for short key duplication
|
||||
if (short != None):
|
||||
if (short in short_map):
|
||||
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
|
||||
else:
|
||||
short_map.append(short)
|
||||
|
||||
# Check for long key duplication
|
||||
if (long in long_map):
|
||||
raise Exception("Long config '{0}' has duplicate entry!".format(long))
|
||||
else:
|
||||
long_map.append(long)
|
||||
|
||||
# Check for long key duplication
|
||||
if (long_invert != None):
|
||||
if (long_invert in long_map):
|
||||
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
|
||||
else:
|
||||
long_map.append(long_invert)
|
||||
|
||||
if (len(sys.argv) < 5):
|
||||
sys.exit()
|
||||
|
||||
@@ -568,8 +414,6 @@ json_object = json.loads(json_text)
|
||||
options = json_object["Options"]
|
||||
unnamed_options = json_object["UnnamedOptions"]
|
||||
|
||||
check_for_duplicate_options(options)
|
||||
|
||||
# Generate config include file
|
||||
output_file = open(output_filename, "w")
|
||||
print_header()
|
||||
@@ -581,7 +425,6 @@ output_file.close()
|
||||
# Generate man file
|
||||
output_man = open(output_man_page, "w")
|
||||
print_man_header()
|
||||
print_man_options(options)
|
||||
print_man_environment(options)
|
||||
print_man_tail()
|
||||
|
||||
@@ -589,8 +432,6 @@ output_man.close()
|
||||
|
||||
# Generate argument loader code
|
||||
output_argloader = open(output_argumentloader_filename, "w")
|
||||
print_argloader_options(options);
|
||||
print_parse_argloader_options(options);
|
||||
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
@@ -58,9 +58,10 @@ class OpDefinition:
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
Inline: list[str]
|
||||
Arguments: list[OpArgument]
|
||||
EmitValidation: list[str]
|
||||
Desc: list[str]
|
||||
|
||||
def __init__(self):
|
||||
self.Name = None
|
||||
@@ -91,19 +92,14 @@ class OpDefinition:
|
||||
attrs = vars(self)
|
||||
print(", ".join("%s: %s" % item for item in attrs.items()))
|
||||
|
||||
IRTypesToCXX = {}
|
||||
CXXTypeToIR = {}
|
||||
IROps = []
|
||||
IRTypesToCXX: dict[str, IRType] = {}
|
||||
CXXTypeToIR: dict[str, IRType] = {}
|
||||
IROps: list[OpDefinition] = []
|
||||
|
||||
IROpNameMap = {}
|
||||
IROpNameSet: set[str] = set()
|
||||
|
||||
def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR"):
|
||||
return True
|
||||
return False
|
||||
def is_ssa_type(op_type: str):
|
||||
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
|
||||
|
||||
def parse_irtypes(irtypes):
|
||||
for op_key, op_val in irtypes.items():
|
||||
@@ -218,11 +214,8 @@ def parse_ops(ops):
|
||||
OpArg.DefaultInitializer = DefaultInit[1][:-1]
|
||||
|
||||
# If SSA type then we can generate validation for this op
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
OpArg.NameWithPrefix = NameWithPrefix
|
||||
@@ -278,6 +271,12 @@ def parse_ops(ops):
|
||||
if "TiedSource" in op_val:
|
||||
OpDef.TiedSource = op_val["TiedSource"]
|
||||
|
||||
# Pad Inline out to the argument count
|
||||
OpDef.Inline = [''] * len(OpDef.Arguments)
|
||||
if "Inline" in op_val:
|
||||
Value = op_val["Inline"]
|
||||
OpDef.Inline[0:len(Value)] = Value
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -289,22 +288,29 @@ def parse_ops(ops):
|
||||
#OpDef.print()
|
||||
|
||||
# Error on duplicate op
|
||||
if OpDef.Name in IROpNameMap:
|
||||
if OpDef.Name in IROpNameSet:
|
||||
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
|
||||
|
||||
IROps.append(OpDef)
|
||||
IROpNameMap[OpDef.Name] = 1
|
||||
IROpNameSet.add(OpDef.Name)
|
||||
|
||||
# Print out enum values
|
||||
def print_enums():
|
||||
def print_enums(enums):
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
for name, members in enums.items():
|
||||
output_file.write(f"enum {name} {{\n")
|
||||
for member in members:
|
||||
if member:
|
||||
output_file.write(f"\t{member}\n")
|
||||
else:
|
||||
output_file.write("\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ENUM\n")
|
||||
output_file.write("#endif\n\n")
|
||||
|
||||
@@ -397,16 +403,16 @@ def print_ir_sizes():
|
||||
// Make sure our array maps directly to the IROps enum
|
||||
static_assert(IRSizes[IROps::OP_LAST] == -1ULL);
|
||||
|
||||
[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool LoweredX87(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);
|
||||
[[nodiscard]] inline size_t GetSize(IROps Op) { return IRSizes[Op]; }
|
||||
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool LoweredX87(IROps Op);
|
||||
[[nodiscard, gnu::const]] int8_t TiedSource(IROps Op);
|
||||
|
||||
#undef IROP_SIZES
|
||||
#endif
|
||||
@@ -415,30 +421,29 @@ def print_ir_sizes():
|
||||
def print_ir_reg_classes():
|
||||
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
for op in IROps:
|
||||
if op.Name == "Last":
|
||||
output_file.write("\tFEXCore::IR::InvalidClass,\n")
|
||||
output_file.write("\tRegClass::Invalid,\n")
|
||||
else:
|
||||
Class = "Invalid"
|
||||
if op.HasDest and op.DestType == None:
|
||||
if op.HasDest and op.DestType is None:
|
||||
ExitError("IR op {} has destination with no destination class".format(op.Name))
|
||||
|
||||
if op.HasDest and op.DestType == "SSA": # Special case SSA type
|
||||
output_file.write("\tFEXCore::IR::ComplexClass,\n")
|
||||
output_file.write("\tRegClass::Complex,\n")
|
||||
elif op.HasDest:
|
||||
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
|
||||
output_file.write("\tRegClass::{},\n".format(op.DestType))
|
||||
else:
|
||||
# No destination so it has an invalid destination class
|
||||
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
|
||||
output_file.write("\tRegClass::Invalid, // No destination\n")
|
||||
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("// Make sure our array maps directly to the IROps enum\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
|
||||
|
||||
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
|
||||
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -561,9 +566,7 @@ def print_ir_arg_printer():
|
||||
|
||||
SSAArgNum = 0
|
||||
FirstArg = True
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
|
||||
for arg in op.Arguments:
|
||||
# No point printing temporaries that we can't recover
|
||||
if arg.Temporary:
|
||||
continue
|
||||
@@ -588,13 +591,13 @@ def print_ir_arg_printer():
|
||||
output_file.write("#endif\n")
|
||||
|
||||
def print_validation(op):
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
if len(op.EmitValidation) != 0:
|
||||
output_file.write("#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("#endif\n")
|
||||
|
||||
# Print out IR allocator helpers
|
||||
def print_ir_allocator_helpers():
|
||||
@@ -664,7 +667,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\treturn HeaderOp->Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
@@ -678,22 +681,21 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
# Output SSA args first
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
if arg.DefaultInitializer:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -751,20 +753,19 @@ def print_ir_allocator_helpers():
|
||||
if op.SSAArgNum:
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
if arg.DefaultInitializer:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -773,9 +774,29 @@ def print_ir_allocator_helpers():
|
||||
output_file.write(") {\n")
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
|
||||
idx = 0
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
# Inline an immediate if we can
|
||||
inline = op.Inline[idx]
|
||||
idx += 1
|
||||
|
||||
if inline != '':
|
||||
Sized = "Size" in [x.Name for x in op.Arguments]
|
||||
P = ["Size" if Sized else "OpSize::i64Bit", arg.Name]
|
||||
|
||||
# A few cases need extra info plumbed.
|
||||
if inline == "SubtractZero":
|
||||
P += ["Src2"]
|
||||
elif inline == "Mem":
|
||||
P += ["OffsetType", "OffsetScale"]
|
||||
elif inline == "Memtso":
|
||||
P += ["OffsetType", "OffsetScale", "true /* TSO */"]
|
||||
inline = "Mem"
|
||||
|
||||
output_file.write(f"\t\t{arg.Name} = Inline{inline}({', '.join(P)});\n")
|
||||
|
||||
output_file.write(f"\t\t{arg.Name}->AddUse();\n")
|
||||
|
||||
# Insert validation here. This is skipped for the
|
||||
# OrderedNodeWrapper version because validation can depend on
|
||||
@@ -785,16 +806,15 @@ def print_ir_allocator_helpers():
|
||||
print_validation(op)
|
||||
|
||||
output_file.write(f"\t\treturn _{op.Name}(")
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
output_file.write(arg.Name)
|
||||
if arg.IsSSA:
|
||||
output_file.write("->Wrapped(ListDataBegin)")
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
output_file.write(");\n");
|
||||
output_file.write("\t}\n\n");
|
||||
output_file.write(");\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
@@ -825,8 +845,8 @@ def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
if len(sys.argv) < 4:
|
||||
ExitError("Insufficient parameters passed to script")
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
@@ -838,6 +858,7 @@ json_file.close()
|
||||
json_object = json.loads(json_text)
|
||||
json_object = {k.upper(): v for k, v in json_object.items()}
|
||||
|
||||
enums = json_object["ENUMS"]
|
||||
ops = json_object["OPS"]
|
||||
irtypes = json_object["IRTYPES"]
|
||||
defines = json_object["DEFINES"]
|
||||
@@ -847,7 +868,7 @@ parse_ops(ops)
|
||||
|
||||
output_file = open(output_filename, "w")
|
||||
|
||||
print_enums()
|
||||
print_enums(enums)
|
||||
print_ir_structs(defines)
|
||||
print_ir_sizes()
|
||||
print_ir_reg_classes()
|
||||
|
||||
@@ -18,14 +18,12 @@ set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/CodeCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/AVX_128.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
@@ -33,7 +31,6 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
@@ -61,16 +58,15 @@ set (SRCS
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
Utils/LongJump.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
@@ -207,7 +203,7 @@ add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
|
||||
@@ -1,20 +1,19 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <memory>
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
constexpr static size_t BUFFER_SIZE = FEXCore::Utils::FEX_PAGE_SIZE - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
@@ -29,7 +28,7 @@ struct JITSymbolBuffer {
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
static_assert(sizeof(JITSymbolBuffer) == FEXCore::Utils::FEX_PAGE_SIZE, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
|
||||
@@ -4,9 +4,9 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
#include "cephes_128bit.h"
|
||||
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
@@ -294,7 +294,11 @@ struct FEX_PACKED X80SoftFloat {
|
||||
FCMP(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
|
||||
*eq = extF80_eq(state, lhs, rhs);
|
||||
*lt = extF80_lt(state, lhs, rhs);
|
||||
*nan = IsNan(lhs) || IsNan(rhs);
|
||||
|
||||
// Use IEEE 754 semantics: unordered if neither <, =, nor > is true
|
||||
// This is more reliable than custom NaN detection
|
||||
bool gt = !(*eq) && !(*lt) && extF80_le(state, rhs, lhs);
|
||||
*nan = !(*eq) && !(*lt) && !gt;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
@@ -497,12 +501,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
float ToF32(softfloat_state* state) const {
|
||||
const float32_t Result = extF80_to_f32(state, *this);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
return std::bit_cast<float>(Result);
|
||||
}
|
||||
|
||||
double ToF64(softfloat_state* state) const {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
return std::bit_cast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
@@ -514,7 +518,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
BIGFLOAT ToFMax(softfloat_state* state) const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(state, *this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
return std::bit_cast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result {};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
@@ -573,18 +577,18 @@ struct FEX_PACKED X80SoftFloat {
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const float rhs) {
|
||||
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
|
||||
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const double rhs) {
|
||||
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
|
||||
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
|
||||
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
|
||||
#else
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
*this = std::bit_cast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -2,74 +2,27 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <concepts>
|
||||
#include <string_view>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, bool* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
template<std::integral T>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
if constexpr (std::is_signed_v<T>) {
|
||||
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
|
||||
} else {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int32_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, fextl::string* Result) {
|
||||
inline bool Conv(std::string_view Value, fextl::string* Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#include <immintrin.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
@@ -13,22 +12,10 @@
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"ShortArg": "n",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
"TextDefault": "none",
|
||||
"Choices": [ "none", "read", "readwrite" ],
|
||||
"ArgumentHandler": "CacheObjectCodeHandler",
|
||||
"Desc": [
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
},
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
@@ -70,7 +57,11 @@
|
||||
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
|
||||
"DISABLEPRESERVEALLABI": "disablepreserveallabi",
|
||||
"ENABLEWFXT": "enablewfxt",
|
||||
"DISABLEWFXT": "disablewfxt"
|
||||
"DISABLEWFXT": "disablewfxt",
|
||||
"ENABLE3DNOW": "enable3dnow",
|
||||
"DISABLE3DNOW": "disable3dnow",
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -92,7 +83,9 @@
|
||||
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
|
||||
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it"
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
@@ -107,7 +100,6 @@
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "R",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
@@ -122,7 +114,6 @@
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
@@ -130,7 +121,6 @@
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
@@ -138,7 +128,6 @@
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "k",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
@@ -153,7 +142,6 @@
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "E",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
@@ -161,7 +149,6 @@
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "H",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
@@ -174,13 +161,50 @@
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
},
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheIncreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "250",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
|
||||
"Lower numbers means more aggressive scaling upward to the maximum size.",
|
||||
"Higher numbers means more conservative scaling, using less memory.",
|
||||
"Can potentially introduce stutters, more likely the higher the number.",
|
||||
"Don't have this number smaller than the decrease count!"
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheDecreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "50",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
|
||||
"The higher the number, the more aggressively it reduces the L1 cache size.",
|
||||
"Lower numbers means more conservative memory savings.",
|
||||
"Can potentially introduce more stutters, more likely the higher the number.",
|
||||
"Don't have this number larger than the increase count!"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "S",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
@@ -188,7 +212,6 @@
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "G",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
@@ -222,7 +245,6 @@
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "g",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
@@ -230,7 +252,6 @@
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "O0",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
@@ -320,7 +341,6 @@
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "s",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
@@ -328,7 +348,6 @@
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
@@ -349,6 +368,13 @@
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
},
|
||||
"EnableGpuvisProfiling": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables profiling when FEX was built with the gpuvis profiler backend."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
@@ -403,20 +429,12 @@
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
@@ -426,23 +444,6 @@
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around flags.",
|
||||
"Assumes flags are not used across cals.",
|
||||
"Hand-written assembly can violate this assumption."
|
||||
]
|
||||
},
|
||||
"ParanoidTSO": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Makes TSO operations even more strict.",
|
||||
"Forces vector loadstores to also become atomic."
|
||||
]
|
||||
},
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -473,32 +474,16 @@
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
},
|
||||
"MonoHacks": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
"AOTIRCapture": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Captures IR and generates an AOT IR cache.",
|
||||
"Captures both the loaded executable and libraries it loads."
|
||||
]
|
||||
},
|
||||
"AOTIRGenerate": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Scans file for executable code and generates an AOT IR cache.",
|
||||
"Does not run the executable."
|
||||
]
|
||||
},
|
||||
"AOTIRLoad": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Loads an AOT IR cache for the loaded executable."
|
||||
]
|
||||
},
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
@@ -512,15 +497,32 @@
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
},
|
||||
"ExtendedVolatileMetadata": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
|
||||
"Limited in its use but can be handy.",
|
||||
"Extends on top of what Microsoft has for volatile metadata, but also supported for WoW64.",
|
||||
"Colon delimited modules, then semi-colon delimited instructions, then comma delimited ranges",
|
||||
"Default disables TSO in the module, unless instructions overlap the range",
|
||||
"<module>;<offset begin>-<offset-end>,...;<instruction offset to force TSO>,...:<another>",
|
||||
"examples:",
|
||||
" * Disable TSO for a full module: Just provide the module name:",
|
||||
" `hl2_linux`",
|
||||
" * Disable TSO for a part of the module:",
|
||||
" `hl2_linux;<offset begin>-<offset-end>`",
|
||||
" * Disable TSO for a part of the module, but enable TSO for some instructions within the module",
|
||||
" `hl2_linux;<offset begin>-<offset-end>;<instruction offset>,<instruction offset>`",
|
||||
" * Disable TSO for multiple modules",
|
||||
" `hl2_linux:libsdl2.so`"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"UnnamedOptions": {
|
||||
"Misc": {
|
||||
"IS_INTERPRETER": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -8,18 +9,12 @@
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
|
||||
}
|
||||
|
||||
@@ -2,61 +2,50 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class SignalDelegator;
|
||||
class ThunkHandler;
|
||||
struct LookupCacheWriteLockToken;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
}
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
struct InternalThreadState;
|
||||
} // namespace Core
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class Dispatcher;
|
||||
} // namespace CPU
|
||||
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
class SyscallHandler;
|
||||
} // namespace HLE
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct IRListCopy;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
class IRValidation;
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct FEX_PACKED ExitFunctionLinkData {
|
||||
uint64_t HostCode;
|
||||
@@ -76,7 +65,23 @@ struct CustomIRResult {
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
public:
|
||||
CodeCache(ContextImpl&);
|
||||
~CodeCache();
|
||||
|
||||
ContextImpl& CTX;
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
void InitiateCacheGeneration() override {
|
||||
IsGeneratingCache = true;
|
||||
}
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
@@ -90,6 +95,7 @@ public:
|
||||
|
||||
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
|
||||
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
uint64_t GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
@@ -145,24 +151,8 @@ public:
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
|
||||
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
CodeCache& GetCodeCache() override {
|
||||
return CodeCache;
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
@@ -173,8 +163,6 @@ public:
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
@@ -189,14 +177,13 @@ public:
|
||||
|
||||
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
|
||||
|
||||
void MarkMonoDetected() override {
|
||||
MonoDetected = true;
|
||||
}
|
||||
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
@@ -209,13 +196,8 @@ public:
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
@@ -223,13 +205,12 @@ public:
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
|
||||
} Config;
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
@@ -243,29 +224,23 @@ public:
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
CodeCache CodeCache;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl(const FEXCore::HostFeatures& Features);
|
||||
~ContextImpl();
|
||||
|
||||
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, const FEXCore::LookupCacheWriteLockToken& lk);
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
// NOTE: Other threads sharing the same CodeBuffer may reference
|
||||
// invalidated data ranges through their L1/L2 caches. This is
|
||||
// not currently a problem since FEX does not repurpose the
|
||||
// invalidated CodeBuffer memory range currently.
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
|
||||
// (safe as the invalidation mutex is locked) and manually invalidates the modified range. Allowing SMC to be detected
|
||||
// even if faulting is disabled.
|
||||
static void MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
void RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint);
|
||||
|
||||
struct GenerateIRResult {
|
||||
std::optional<IR::IRListView> IRView;
|
||||
@@ -292,9 +267,9 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator {"FEXMem_OpDispatcher"};
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator {"FEXMem_Frontend"};
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator {"FEXMem_CPUBackend"};
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
@@ -324,6 +299,10 @@ public:
|
||||
return ExitOnHLT;
|
||||
}
|
||||
|
||||
bool AreMonoHacksActive() const {
|
||||
return Config.MonoHacks && MonoDetected;
|
||||
}
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
@@ -331,17 +310,10 @@ protected:
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else if (Config.ParanoidTSO) {
|
||||
AtomicTSOEmulationEnabled = true;
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
|
||||
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
|
||||
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
AtomicTSOEmulationEnabled = Config.TSOEnabled;
|
||||
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -355,10 +327,6 @@ private:
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool VectorAtomicTSOEmulationEnabled = false;
|
||||
@@ -371,11 +339,14 @@ private:
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
struct CustomIRHandlerEntry final {
|
||||
CustomIREntrypointHandler Handler;
|
||||
void *Creator;
|
||||
void *Data;
|
||||
void* Creator;
|
||||
void* Data;
|
||||
};
|
||||
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
|
||||
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
|
||||
fextl::set<uint64_t> ForceTSOInstructions;
|
||||
|
||||
bool MonoDetected = false;
|
||||
std::atomic<uint64_t> MonoBackpatcherBlock;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -7,12 +7,11 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = IREmit->Constant(A.Offset);
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
@@ -25,7 +24,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,27 +45,23 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
};
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize) {
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
}
|
||||
|
||||
// Loadstore rules:
|
||||
@@ -100,7 +95,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
|
||||
if ((AtomicTSO && !Vector && HostSupportsTSOImm9 && OffsetIsSIMM9) || (!AtomicTSO && (OffsetIsSIMM9 || OffsetIsUnsignedScaled))) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
@@ -108,35 +103,25 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
|
||||
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
}
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
AddressMode B = A;
|
||||
|
||||
// ScaledRegisterLoadstore
|
||||
if (B.Index && B.Segment) {
|
||||
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
|
||||
} else if (B.Segment) {
|
||||
B.Index = B.Segment;
|
||||
B.IndexScale = 1;
|
||||
}
|
||||
|
||||
return B;
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
@@ -151,7 +136,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
@@ -159,7 +144,10 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return SoftwareAddressCalculation();
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -11,17 +11,18 @@ struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
MemOffsetType IndexType = MemOffsetType::SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize);
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,10 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
@@ -712,7 +712,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
|
||||
|
||||
|
||||
@@ -1,30 +1,31 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#endif
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::X86State {
|
||||
enum X86Reg : uint32_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Contains the address to the currently available CPU state
|
||||
|
||||
@@ -1,14 +1,17 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <linux/prctl.h>
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
|
||||
@@ -317,7 +320,7 @@ namespace CPU {
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
|
||||
|
||||
RegisterForSignalHandler(PrevCodeBuffer);
|
||||
RegisterForSignalHandler(std::move(PrevCodeBuffer));
|
||||
return CurrentCodeBuffer.get();
|
||||
}
|
||||
|
||||
@@ -326,14 +329,13 @@ namespace CPU {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Keep a reference to the old code buffer to delay deallocation
|
||||
SignalHandlerCodeBuffers.push_back(CodeBuffer);
|
||||
SignalHandlerCodeBuffers.push_back(std::move(CodeBuffer));
|
||||
} else {
|
||||
SignalHandlerCodeBuffers.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
|
||||
auto NewCodeBuffer = CodeBuffers.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
@@ -358,6 +360,8 @@ namespace CPU {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
|
||||
@@ -17,6 +17,10 @@ $end_info$
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
}
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace IR {
|
||||
@@ -157,18 +161,7 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
*
|
||||
* @param Entry - RIP of the entry
|
||||
* @param SerializationData - Serialization data referring to the object cache for `Entry`
|
||||
*
|
||||
* @return An executable function pointer relocated from the cache object
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
|
||||
return nullptr;
|
||||
}
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
|
||||
@@ -43,12 +43,15 @@ namespace ProductNames {
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_A725[] = "Cortex-A725";
|
||||
static const char ARM_C1Pro[] = "C1-Pro";
|
||||
static const char ARM_C1Premium[] = "C1-Premium";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_X925[] = "Cortex-X925";
|
||||
static const char ARM_C1Ultra[] = "C1-Ultra";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_N3[] = "Neoverse N3";
|
||||
@@ -59,6 +62,7 @@ namespace ProductNames {
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
static const char ARM_A520[] = "Cortex-A520";
|
||||
static const char ARM_C1Nano[] = "C1-Nano";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
@@ -70,6 +74,7 @@ namespace ProductNames {
|
||||
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
static const char ARM_Olympus[] = "Nvidia Olympus";
|
||||
|
||||
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
|
||||
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
|
||||
@@ -83,8 +88,12 @@ namespace ProductNames {
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
static const char ARM_AppleSilicon[] = "Apple Silicon";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
static const char ARM_Ampere_1[] = "AmpereOne";
|
||||
static const char ARM_Ampere_1A[] = "AmpereOneA";
|
||||
static const char ARM_Ampere_1B[] = "AmpereOneB";
|
||||
#else
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
@@ -170,7 +179,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
@@ -180,39 +189,48 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
{0x61, 0, 1, ProductNames::ARM_AppleSilicon}, // QEmu Apple Silicon
|
||||
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
|
||||
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
|
||||
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
|
||||
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
|
||||
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
|
||||
|
||||
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
|
||||
// Denver rated above A57 to match TX2 weirdness
|
||||
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
|
||||
@@ -227,6 +245,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
|
||||
|
||||
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
@@ -892,71 +911,71 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(0 << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extensions
|
||||
(1 << 4) | // TSC
|
||||
(1 << 5) | // MSR support
|
||||
(1 << 6) | // PAE
|
||||
(1 << 7) | // Machine Check Exception
|
||||
(1 << 8) | // CMPXCHG8B
|
||||
(1 << 9) | // APIC
|
||||
(0 << 10) | // Reserved
|
||||
(1 << 11) | // SYSCALL/SYSRET
|
||||
(1 << 12) | // MTRR
|
||||
(1 << 13) | // Page global extension
|
||||
(1 << 14) | // Machine Check architecture
|
||||
(1 << 15) | // CMOV
|
||||
(1 << 16) | // Page attribute table
|
||||
(1 << 17) | // Page-size extensions
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(1 << 20) | // NX
|
||||
(0 << 21) | // Reserved
|
||||
(1 << 22) | // MMXExt
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(1 << 30) | // 3DNow! Extensions
|
||||
(1 << 31); // 3DNow!
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extensions
|
||||
(1 << 4) | // TSC
|
||||
(1 << 5) | // MSR support
|
||||
(1 << 6) | // PAE
|
||||
(1 << 7) | // Machine Check Exception
|
||||
(1 << 8) | // CMPXCHG8B
|
||||
(1 << 9) | // APIC
|
||||
(0 << 10) | // Reserved
|
||||
(1 << 11) | // SYSCALL/SYSRET
|
||||
(1 << 12) | // MTRR
|
||||
(1 << 13) | // Page global extension
|
||||
(1 << 14) | // Machine Check architecture
|
||||
(1 << 15) | // CMOV
|
||||
(1 << 16) | // Page attribute table
|
||||
(1 << 17) | // Page-size extensions
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(1 << 20) | // NX
|
||||
(0 << 21) | // Reserved
|
||||
(1 << 22) | // MMXExt
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(CTX->HostFeatures.Supports3DNow << 30) | // 3DNow! Extensions
|
||||
(CTX->HostFeatures.Supports3DNow << 31); // 3DNow!
|
||||
return Res;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <Interface/Context/Context.h>
|
||||
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
ExecutableFileInfo::~ExecutableFileInfo() = default;
|
||||
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Context {
|
||||
|
||||
CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
|
||||
// TODO
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -14,11 +14,11 @@ $end_info$
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include <Interface/GDBJIT/GDBJIT.h>
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -78,10 +78,7 @@ namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
: HostFeatures {Features}
|
||||
, CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
, CodeCache {*this} {
|
||||
if (!Config.Is64BitMode()) {
|
||||
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
|
||||
Config.VirtualMemSize = 1ULL << 32;
|
||||
@@ -105,14 +102,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
CodeObjectCacheService->Shutdown();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
|
||||
const CPU::CPUBackend::JITCodeTail* InlineTail;
|
||||
@@ -139,6 +128,11 @@ bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* T
|
||||
return InlineTail && InlineTail->SingleInst;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail ? InlineTail->RIP : 0;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
@@ -349,36 +343,9 @@ bool ContextImpl::InitCore() {
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
|
||||
|
||||
.AbsoluteLoopTopAddress = Dispatcher->AbsoluteLoopTopAddress,
|
||||
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
|
||||
|
||||
.PauseReturnInstruction = Dispatcher->PauseReturnInstruction,
|
||||
.ThreadPauseHandlerAddressSpillSRA = Dispatcher->ThreadPauseHandlerAddressSpillSRA,
|
||||
.ThreadPauseHandlerAddress = Dispatcher->ThreadPauseHandlerAddress,
|
||||
|
||||
// Stop handlers.
|
||||
.ThreadStopHandlerAddressSpillSRA = Dispatcher->ThreadStopHandlerAddressSpillSRA,
|
||||
.ThreadStopHandlerAddress = Dispatcher->ThreadStopHandlerAddress,
|
||||
|
||||
// SRA information.
|
||||
.SRAGPRCount = Dispatcher->GetSRAGPRCount(),
|
||||
.SRAFPRCount = Dispatcher->GetSRAFPRCount(),
|
||||
};
|
||||
|
||||
Dispatcher->GetSRAGPRMapping(SignalConfig.SRAGPRMapping);
|
||||
Dispatcher->GetSRAFPRMapping(SignalConfig.SRAFPRMapping);
|
||||
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
#elif !defined(_M_ARM_64EC)
|
||||
#if defined(_WIN32) && !defined(_M_ARM_64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
@@ -398,12 +365,6 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
// TODO: This doesn't make sense when the parent thread doesn't outlive its children
|
||||
}
|
||||
@@ -415,7 +376,9 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
@@ -437,26 +400,11 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Set up default code segment.
|
||||
// Default code segment indexes match the numbers that the Linux kernel uses.
|
||||
Thread->CurrentFrame->State.cs_idx = 6 << 3;
|
||||
auto &GDT = Thread->CurrentFrame->State.gdt[Thread->CurrentFrame->State.cs_idx >> 3];
|
||||
Thread->CurrentFrame->State.SetGDTBase(&GDT, 0);
|
||||
Thread->CurrentFrame->State.SetGDTLimit(&GDT, 0xF'FFFFU);
|
||||
|
||||
if (Config.Is64BitMode) {
|
||||
GDT.L = 1; // L = Long Mode = 64-bit
|
||||
GDT.D = 0; // D = Default Operand SIze = Reserved
|
||||
}
|
||||
else {
|
||||
GDT.L = 0; // L = Long Mode = 32-bit
|
||||
GDT.D = 1; // D = Default Operand Size = 32-bit
|
||||
}
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
@@ -520,18 +468,13 @@ void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
if (NewCodeBuffer) {
|
||||
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
Thread->LookupCache->ClearCache();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
Thread->LookupCache->ClearCache(lk);
|
||||
}
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
@@ -579,7 +522,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
auto CodeBlocks = &BlockInfo->Blocks;
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode);
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
|
||||
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
|
||||
|
||||
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
|
||||
|
||||
@@ -607,7 +551,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
@@ -637,11 +581,12 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1],
|
||||
(uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
|
||||
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
std::array<uint8_t, 0x10> CodeOriginal;
|
||||
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
|
||||
|
||||
@@ -651,7 +596,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -659,15 +604,21 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
}
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher.OpDispatch;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
IR::ForceTSOMode ForceTSO =
|
||||
BlockInForceTSOValidRange ?
|
||||
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
|
||||
IR::ForceTSOMode::ForceDisabled) :
|
||||
IR::ForceTSOMode::NoOverride;
|
||||
IR::ForceTSOMode ForceTSO = IR::ForceTSOMode::NoOverride;
|
||||
if (BlockInForceTSOValidRange) {
|
||||
if (InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress) {
|
||||
ForceTSO = IR::ForceTSOMode::ForceEnabled;
|
||||
} else {
|
||||
ForceTSO = IR::ForceTSOMode::ForceDisabled;
|
||||
}
|
||||
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
|
||||
ForceTSO = IR::ForceTSOMode::ForceEnabled;
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->SetForceTSO(ForceTSO);
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
@@ -695,10 +646,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::NOEXEC_INST) {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
} else {
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST) {
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
} else {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -717,7 +668,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -760,27 +712,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
// JIT Code object cache lookup
|
||||
if (CodeObjectCacheService) {
|
||||
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
|
||||
if (CodeCacheEntry) {
|
||||
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = {},
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
.NeedsAddGuestCodeRanges = false,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -797,7 +732,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// but this would increase lock contention. Redundant frontend runs aren't
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
@@ -838,10 +773,13 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
|
||||
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
@@ -855,51 +793,44 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = CompiledCode.BlockBegin;
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
|
||||
GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
|
||||
GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
|
||||
GuestRIP - GuestRIPLookup->FileStartVA);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
|
||||
GuestRIP - GuestRIPLookup->FileStartVA);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Tell the object cache service to serialize the code if enabled
|
||||
if (CodeObjectCacheService && Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE && DebugData) {
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(
|
||||
fextl::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(CodeSerialize::AsyncJobHandler::SerializationJobData {
|
||||
.GuestRIP = GuestRIP,
|
||||
.GuestCodeLength = Length,
|
||||
.GuestCodeHash = 0,
|
||||
.HostCodeBegin = CompiledCode.BlockBegin,
|
||||
.HostCodeLength = CompiledCode.Size,
|
||||
.HostCodeHash = 0,
|
||||
.ThreadJobRefCount = &Thread->ObjectCacheRefCounter,
|
||||
.Relocations = std::move(*DebugData->Relocations),
|
||||
}));
|
||||
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
if (Config.LibraryJITNaming()) {
|
||||
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
|
||||
}
|
||||
|
||||
if (Config.GDBSymbols()) {
|
||||
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
@@ -907,7 +838,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
// contain code, inform the frontend so it can setup SMC detection.
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
for (auto CodePage : BlockInfo->CodePages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
@@ -915,7 +846,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Insert to lookup cache
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
|
||||
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, HostAddr);
|
||||
}
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
@@ -948,7 +879,7 @@ static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
|
||||
Thread->FrontendDecoder->ResetExecutableRangeCache();
|
||||
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
auto& CodePages = Thread->LookupCache->Shared->CodePages;
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
@@ -961,7 +892,7 @@ static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
bool InvalidatedAnyEntries = false;
|
||||
for (const auto& PageEntries : Accumulator) {
|
||||
for (const auto& Entry : PageEntries) {
|
||||
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry)) {
|
||||
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry, lk)) {
|
||||
InvalidatedAnyEntries = true;
|
||||
}
|
||||
}
|
||||
@@ -978,28 +909,16 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
if (!Thread) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
|
||||
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP,
|
||||
const FEXCore::LookupCacheWriteLockToken& lk) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP, lk);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
|
||||
}
|
||||
|
||||
std::optional<CustomIRResult>
|
||||
@@ -1041,10 +960,10 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
|
||||
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
|
||||
},
|
||||
@@ -1065,7 +984,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
ForceTSOInstructions.merge(std::move(Instructions));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
@@ -1075,25 +994,36 @@ void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
void ContextImpl::MarkMonoBackpatcherBlock(uint64_t BlockEntry) {
|
||||
MonoBackpatcherBlock.store(BlockEntry, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidatedEntryAccumulator Accumulator;
|
||||
InvalidateGuestCodeRange(nullptr, Accumulator, Entrypoint, 1);
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
|
||||
HasCustomIRHandlers = !CustomIRHandlers.empty();
|
||||
SyscallHandler->InvalidateGuestCodeRange(Thread, Entrypoint, 1);
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
return rv;
|
||||
}
|
||||
void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
|
||||
{
|
||||
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (Size == 8) {
|
||||
*reinterpret_cast<uint64_t*>(Address) = Value;
|
||||
} else if (Size == 4) {
|
||||
*reinterpret_cast<uint32_t*>(Address) = Value;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
|
||||
}
|
||||
}
|
||||
|
||||
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/VectorRegType.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
@@ -16,14 +17,18 @@
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <csignal>
|
||||
#include <cstring>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -31,12 +36,14 @@ static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuSt
|
||||
CTX->SyscallHandler->SleepThread(CTX, Frame);
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 4;
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = FEXCore::Utils::FEX_PAGE_SIZE * 4;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx} {
|
||||
EmitDispatcher();
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(GetBufferBase()), MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
Dispatcher::~Dispatcher() {
|
||||
@@ -86,12 +93,12 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
@@ -99,10 +106,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
|
||||
// Load ThreadState and write the target PC there
|
||||
@@ -123,7 +130,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, REG_CALLRET_SP);
|
||||
// EC_CALL_CHECKER_PC_REG is REG_PF which isn't touched by any of the above
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, EC_CALL_CHECKER_PC_REG, TMP1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
|
||||
|
||||
// If the entry at the TOS is for the target address, pop it and return to the JIT code
|
||||
add(ARMEmitter::Size::i64Bit, REG_CALLRET_SP, REG_CALLRET_SP, 0x10);
|
||||
@@ -134,86 +141,101 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
ARMEmitter::BiDirectionalLabel FullLookup {};
|
||||
ARMEmitter::BiDirectionalLabel CallBlock {};
|
||||
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::ForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
|
||||
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
tbz(TMP1, 0, &l_NotECCode);
|
||||
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
(void)Bind(&l_NotECCode);
|
||||
#endif
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
|
||||
br(TMP4);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
Bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
if (DisableL2Cache()) {
|
||||
(void)b(&NoBlock);
|
||||
} else {
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Load the pointer from the offset
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
// Load the pointer from the offset
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
// If page pointer is zero then we have no block
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg.R(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -287,39 +309,9 @@ void Dispatcher::EmitDispatcher() {
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
auto EmitECExitCheck = [&]() {
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::ForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
|
||||
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
tbz(TMP1, 0, &l_NotECCode);
|
||||
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
};
|
||||
#endif
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
(void)Bind(&NoBlock);
|
||||
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -353,11 +345,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
{
|
||||
Bind(&CompileSingleStep);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
(void)Bind(&CompileSingleStep);
|
||||
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -519,7 +507,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
@@ -569,8 +557,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_F64_F64_PTR,
|
||||
FABI_F64_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
@@ -578,7 +566,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_F80x2_I16_F80_PTR,
|
||||
FABI_F64x2_I16_F64_PTR,
|
||||
FABI_F64x2_F64_PTR,
|
||||
FABI_I32_I64_I64_V128_V128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
}};
|
||||
@@ -588,14 +576,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&l_CTX);
|
||||
(void)Bind(&l_CTX);
|
||||
dc64(reinterpret_cast<uintptr_t>(CTX));
|
||||
Bind(&l_Sleep);
|
||||
(void)Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
(void)Bind(&l_CompileBlock);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMFCompileBlock.GetConvertedPointer());
|
||||
Bind(&l_CompileSingleStep);
|
||||
(void)Bind(&l_CompileSingleStep);
|
||||
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
|
||||
dc64(PMFCompileSingleStep.GetConvertedPointer());
|
||||
|
||||
@@ -777,7 +766,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
mov(VABI1.Q(), VTMP1.Q());
|
||||
}
|
||||
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
@@ -810,7 +799,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
case FABI_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -820,18 +809,17 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(FallbackPointerReg);
|
||||
GenerateIndirectRuntimeCall<double, double, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
case FABI_F64_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -844,10 +832,9 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
fmov(VABI2.D(), VTMP2.D());
|
||||
}
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(FallbackPointerReg);
|
||||
GenerateIndirectRuntimeCall<double, double, double, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
@@ -1007,7 +994,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
FillF80x2Result();
|
||||
} break;
|
||||
case FABI_F64x2_I16_F64_PTR: {
|
||||
case FABI_F64x2_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -1016,14 +1003,13 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, uint16_t, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
|
||||
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
@@ -1125,6 +1111,53 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
}
|
||||
|
||||
SignalDelegatorConfig Dispatcher::MakeSignalDelegatorConfig() const {
|
||||
// PF/AF are the final two SRA registers. We only want GPRs
|
||||
const auto GPRCount = uint16_t(StaticRegisters.size() - 2);
|
||||
const auto FPRCount = uint16_t(StaticFPRegisters.size());
|
||||
|
||||
const auto GetSRAGPRMapping = [GPRCount, this] {
|
||||
SignalDelegatorConfig::SRAIndexMapping Mapping {};
|
||||
for (size_t i = 0; i < GPRCount; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
return Mapping;
|
||||
};
|
||||
|
||||
const auto GetSRAFPRMapping = [FPRCount, this] {
|
||||
SignalDelegatorConfig::SRAIndexMapping Mapping {};
|
||||
for (size_t i = 0; i < FPRCount; ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
return Mapping;
|
||||
};
|
||||
|
||||
return FEXCore::SignalDelegatorConfig {
|
||||
.DispatcherBegin = Start,
|
||||
.DispatcherEnd = End,
|
||||
|
||||
.AbsoluteLoopTopAddress = AbsoluteLoopTopAddress,
|
||||
.AbsoluteLoopTopAddressFillSRA = AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = SignalHandlerReturnAddressRT,
|
||||
|
||||
.PauseReturnInstruction = PauseReturnInstruction,
|
||||
.ThreadPauseHandlerAddressSpillSRA = ThreadPauseHandlerAddressSpillSRA,
|
||||
.ThreadPauseHandlerAddress = ThreadPauseHandlerAddress,
|
||||
|
||||
// Stop handlers.
|
||||
.ThreadStopHandlerAddressSpillSRA = ThreadStopHandlerAddressSpillSRA,
|
||||
.ThreadStopHandlerAddress = ThreadStopHandlerAddress,
|
||||
|
||||
// SRA information.
|
||||
.SRAGPRCount = GPRCount,
|
||||
.SRAFPRCount = FPRCount,
|
||||
|
||||
.SRAGPRMapping = GetSRAGPRMapping(),
|
||||
.SRAFPRMapping = GetSRAFPRMapping(),
|
||||
};
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
|
||||
return fextl::make_unique<Dispatcher>(CTX);
|
||||
}
|
||||
|
||||
@@ -2,25 +2,19 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestSigAction;
|
||||
}
|
||||
struct SignalDelegatorConfig;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct CpuStateFrame;
|
||||
@@ -42,6 +36,32 @@ public:
|
||||
Dispatcher(FEXCore::Context::ContextImpl* ctx);
|
||||
~Dispatcher();
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
#else
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
DispatchPtr(Frame, false);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
#endif
|
||||
|
||||
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame, bool SingleInst);
|
||||
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
private:
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -59,68 +79,22 @@ public:
|
||||
uint64_t GuestSignal_SIGILL {};
|
||||
uint64_t GuestSignal_SIGTRAP {};
|
||||
uint64_t GuestSignal_SIGSEGV {};
|
||||
uint64_t IntCallbackReturnAddress {};
|
||||
|
||||
uint64_t PauseReturnInstruction {};
|
||||
std::array<uint64_t, FallbackABI::FABI_UNKNOWN> ABIPointers {};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint64_t Start {};
|
||||
uint64_t End {};
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
#else
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
DispatchPtr(Frame, false);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
#endif
|
||||
|
||||
uint16_t GetSRAGPRCount() const {
|
||||
// PF/AF are the final two SRA registers.
|
||||
// Only return the SRA for GPRs.
|
||||
return StaticRegisters.size() - 2;
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const {
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame, bool SingleInst);
|
||||
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress {};
|
||||
uint64_t LDIVHandlerAddress {};
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -71,7 +71,23 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
: Thread {Thread}
|
||||
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
|
||||
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
if (ReducedPrecision) {
|
||||
X87Table = &FEXCore::X86Tables::X87F64Ops;
|
||||
} else {
|
||||
X87Table = &FEXCore::X86Tables::X87F80Ops;
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
VEXTable = &FEXCore::X86Tables::VEXTableOps;
|
||||
VEXTableGroup = &FEXCore::X86Tables::VEXTableGroupOps;
|
||||
} else if (CTX->HostFeatures.SupportsAVX) {
|
||||
VEXTable = &FEXCore::X86Tables::VEXTableOps_AVX128;
|
||||
VEXTableGroup = &FEXCore::X86Tables::VEXTableGroupOps_AVX128;
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Treat FEX-internal X86 callbacks as always executable
|
||||
@@ -83,6 +99,7 @@ bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
|
||||
ExecutableRangeBase = RangeInfo.Base;
|
||||
ExecutableRangeEnd = RangeInfo.Base + RangeInfo.Size;
|
||||
ExecutableRangeWritable = RangeInfo.Writable;
|
||||
|
||||
if (RangeInfo.Size == 0) {
|
||||
return false;
|
||||
@@ -242,13 +259,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
|
||||
if (HasSIB) {
|
||||
FEXCore::X86Tables::SIBDecoded SIB;
|
||||
if (DecodeInst->DecodedSIB) {
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
} else {
|
||||
// Haven't yet grabbed SIB, pull it now
|
||||
DecodeInst->SIB = ReadByte();
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
DecodeInst->DecodedSIB = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
|
||||
}
|
||||
|
||||
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
|
||||
@@ -314,6 +331,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
}
|
||||
|
||||
bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
|
||||
// Dispatcher Op.
|
||||
// TODO: Move this in to `NormalOpHeader`, Dispatch tables have a bug currently where some subtables don't inherit flags correctly.
|
||||
// Can be seen by running FEX asm tests if this is removed.
|
||||
return NormalOp(&Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0], Op);
|
||||
}
|
||||
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
@@ -377,9 +401,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
// If we require ModRM and haven't decoded it yet, do it now
|
||||
// Some instructions have to read modrm upfront, others do it later
|
||||
if (HasMODRM && !DecodeInst->DecodedModRM) {
|
||||
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
|
||||
DecodeInst->ModRM = ReadByte();
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
}
|
||||
|
||||
// New instruction size decoding
|
||||
@@ -412,9 +436,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
|
||||
DestSize = 2;
|
||||
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) &&
|
||||
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
|
||||
DestSize = 8;
|
||||
} else {
|
||||
@@ -441,9 +464,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
// See table 1-2. Operand-Size Overrides for this decoding
|
||||
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) &&
|
||||
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
|
||||
@@ -612,11 +634,20 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
|
||||
}
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
++CurrentSrc;
|
||||
|
||||
if (Bytes == 8) [[unlikely]] {
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
@@ -626,7 +657,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
}
|
||||
|
||||
bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->OPRaw = DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
@@ -642,10 +673,13 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
// A normal instruction is the most likely.
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INST) [[likely]] {
|
||||
return NormalOp(Info, Op);
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
|
||||
// Dispatcher Op.
|
||||
return NormalOp(&Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0], Op);
|
||||
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -662,24 +696,24 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
uint16_t PrefixType = PF_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0xF3) {
|
||||
if (LastEscapePrefix == 0xF3) {
|
||||
PrefixType = PF_F3;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
|
||||
} else if (LastEscapePrefix == 0xF2) {
|
||||
PrefixType = PF_F2;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66) {
|
||||
} else if (LastEscapePrefix == 0x66) {
|
||||
PrefixType = PF_66;
|
||||
}
|
||||
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
uint16_t LocalOp = OPD(Info->Type, PrefixType, ModRM.reg);
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
const FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
#undef OPD
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM && ModRM.mod == 0b11) {
|
||||
// Everything in this group is privileged instructions aside from XGETBV
|
||||
@@ -700,11 +734,16 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
return NormalOp(&(*X87Table)[X87Op], X87Op);
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
if (!VEXTable) {
|
||||
// AVX not enabled.
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t map_select = 1;
|
||||
uint16_t pp = 0;
|
||||
const uint8_t Byte1 = ReadByte();
|
||||
@@ -742,7 +781,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -752,13 +790,13 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
Op = OPD(map_select, pp, VEXOp);
|
||||
#undef OPD
|
||||
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &VEXTableOps[Op];
|
||||
const FEXCore::X86Tables::X86InstInfo* LocalInfo = &(*VEXTable)[Op];
|
||||
|
||||
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 && LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -766,7 +804,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
|
||||
#undef OPD
|
||||
return NormalOp(&VEXTableGroupOps[Op], Op, options);
|
||||
return NormalOp(&(*VEXTableGroup)[Op], Op, options);
|
||||
} else {
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
}
|
||||
@@ -782,6 +820,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
|
||||
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
InstructionSize = 0;
|
||||
LastEscapePrefix = 0;
|
||||
Instruction.fill(0);
|
||||
|
||||
DecodeInst = &DecodedBuffer[DecodedSize];
|
||||
@@ -803,7 +842,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -842,7 +881,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
|
||||
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather than modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
@@ -858,7 +897,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
constexpr uint16_t PF_3A_REX = (1 << 1);
|
||||
|
||||
uint16_t Prefix = PF_3A_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
|
||||
if (LastEscapePrefix == 0x66) { // Operand Size
|
||||
Prefix = PF_3A_66;
|
||||
}
|
||||
|
||||
@@ -884,17 +923,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
|
||||
} else if (LastEscapePrefix == 0xF3) { // REP
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
|
||||
} else if (LastEscapePrefix == 0xF2) { // REPNE
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
@@ -910,7 +949,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
}
|
||||
case 0x66: // Operand Size prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
break;
|
||||
case 0x67: // Address Size override prefix
|
||||
@@ -941,11 +980,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
break;
|
||||
case 0xF2: // REPNE prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
break;
|
||||
case 0xF3: // REP prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
break;
|
||||
case 0x64: // FS prefix
|
||||
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
|
||||
@@ -955,7 +994,10 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
break;
|
||||
default:
|
||||
[[likely]] { // Default base table
|
||||
auto Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
const X86InstInfo* Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) {
|
||||
Info = &Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0];
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
@@ -1005,13 +1047,32 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
auto Result = ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST :
|
||||
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
|
||||
DecodedBlockStatus::NOEXEC_INST;
|
||||
DecodeInst->InstSize = 0;
|
||||
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
} else if (!DecodeInst->TableInfo || !DecodeInst->TableInfo->OpcodeDispatcher) {
|
||||
return Result;
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (CTX->AreMonoHacksActive()) {
|
||||
// Unity uses a standard SPSC ringbuffer with cached read/write pointers and thread waiting flags at the following
|
||||
// offsets, which are consistent between 32-bit and 64-bit Unity versions from 2015 onwards.
|
||||
auto IsKnownAtomicDisplacement = [](uint64_t Displacement) {
|
||||
return Displacement == 0x80 || Displacement == 0x84 || Displacement == 0xC0 || Displacement == 0xC4;
|
||||
};
|
||||
|
||||
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
|
||||
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
|
||||
}
|
||||
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
|
||||
}
|
||||
}
|
||||
|
||||
return DecodedBlockStatus::SUCCESS;
|
||||
}
|
||||
|
||||
@@ -1027,6 +1088,13 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
const auto InstEnd = DecodeInst->PC + DecodeInst->InstSize;
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_CALL) {
|
||||
if (ExecutableRangeWritable && CTX->AreMonoHacksActive()) {
|
||||
// Mono generated code often contains noreturn calls with garbage following them, and calls are always backpatched
|
||||
// after CIL compilation leading to n recompiles for a multiblock with n calls. Choose to minimize stutters over
|
||||
// raw performance and disable tracking past calls for mono generated code.
|
||||
return;
|
||||
}
|
||||
|
||||
AddBranchTarget(InstEnd);
|
||||
BlockInfo.EntryPoints.emplace(InstEnd);
|
||||
return;
|
||||
@@ -1058,10 +1126,16 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
if (Conditional) {
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
AddBranchTarget(InstEnd);
|
||||
}
|
||||
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
// Forbid cross-page branches to both avoid massive (range-wise) code blocks in highly fragmented code and trying to decode unmapped branch targets
|
||||
bool ValidMultiblockMember =
|
||||
TargetRIP >= SymbolMinAddress && TargetRIP < std::min(FEXCore::AlignUp(InstEnd, FEXCore::Utils::FEX_PAGE_SIZE), SymbolMaxAddress);
|
||||
// Forbid distant branches to have the cost code better match the guest code layout, avoiding massive (range-wise) code
|
||||
// blocks in highly fragmented guest code. Such branches are often not-taken branches to garbage in obfuscated code.
|
||||
constexpr uint64_t MAX_FORWARD_BRANCH_DIST = FEXCore::Utils::FEX_PAGE_SIZE * 4;
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SymbolMaxAddress);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
@@ -1072,9 +1146,6 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
if (Conditional) {
|
||||
MaxCondBranchForward = std::max(MaxCondBranchForward, TargetRIP);
|
||||
MaxCondBranchBackwards = std::min(MaxCondBranchBackwards, TargetRIP);
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
AddBranchTarget(InstEnd);
|
||||
}
|
||||
|
||||
AddBranchTarget(TargetRIP);
|
||||
@@ -1085,6 +1156,60 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::IsBranchMonoTailcall(uint64_t NumInstructions) const {
|
||||
// While the mono call backpatching block can easily be detected due it being the only one to contain SMC-faulting
|
||||
// atomics, that can't be said for the tailcall jump backpatcher which has changed several times across versions and
|
||||
// can be partially inlined. To work around this, instead detect the tailcall site itself and force full non-signal-based
|
||||
// SMC detection for that single block.
|
||||
if (!ExecutableRangeWritable) {
|
||||
// We only care about jitted code
|
||||
return false;
|
||||
}
|
||||
|
||||
// See mini-{amd64,x86}.c in the mono codebase, specifically where METHOD_JUMP patches are emitted.
|
||||
if (GetGPROpSize() == IR::OpSize::i32Bit) {
|
||||
// Matches:
|
||||
// LEAVE
|
||||
// <none> / NOP / MOV EAX, EAX / LEA EBP, [EBP+0]
|
||||
// JMP imm32
|
||||
if (DecodeInst->OP != 0xE9 || NumInstructions < 2) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto PrevInst = std::prev(DecodeInst);
|
||||
if (PrevInst->OP == 0xC9) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (NumInstructions < 3 || std::prev(PrevInst)->OP != 0xC9) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return PrevInst->OP == 0x90 || (PrevInst->OP == 0x8B && PrevInst->ModRM == 0xC0) ||
|
||||
(PrevInst->OP == 0x8D && PrevInst->ModRM == 0x6D && PrevInst->Src[1].IsLiteral() && PrevInst->Src[1].Literal() == 0);
|
||||
} else {
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
if (DecodeInst->OPRaw == 0xFF && ModRM.reg == 4 && DecodeInst->Src[0].IsGPR()) {
|
||||
if (DecodeInst->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// Found in versions of mono from 2024 onwards - matches:
|
||||
// REX.W JMP rax
|
||||
return (DecodeInst->Flags & (DecodeFlags::FLAG_REX_PREFIX | DecodeFlags::FLAG_REX_WIDENING | DecodeFlags::FLAG_REX_XGPR_B |
|
||||
DecodeFlags::FLAG_REX_XGPR_X | DecodeFlags::FLAG_REX_XGPR_R)) ==
|
||||
(DecodeFlags::FLAG_REX_PREFIX | DecodeFlags::FLAG_REX_WIDENING);
|
||||
} else if (NumInstructions > 1 && DecodeInst->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_R11) {
|
||||
// Found in older versions of mono - match:
|
||||
// MOV r11, imm64
|
||||
// JMP r11
|
||||
auto PrevInst = std::prev(DecodeInst);
|
||||
return PrevInst->OP == 0xBB && PrevInst->Dest.IsGPR() && PrevInst->Dest.Data.GPR.GPR == FEXCore::X86State::REG_R11;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Decoder::InstCanContinue() const {
|
||||
if (DecodeInst->PC + DecodeInst->InstSize == NextBlockStartAddress) {
|
||||
return false;
|
||||
@@ -1187,7 +1312,7 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
@@ -1199,8 +1324,8 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thre
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Thread->CurrentFrame->State.gdt[Thread->CurrentFrame->State.cs_idx >> 3];
|
||||
BlockInfo.Is64BitMode = CSSegment.L == 1;
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
// XXX: Load symbol data
|
||||
@@ -1328,7 +1453,10 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thre
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" : "NoExec", OpAddress);
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
|
||||
"PartialDecode",
|
||||
OpAddress);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1345,6 +1473,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thre
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
|
||||
@@ -4,18 +4,21 @@
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <stddef.h>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::HLE {
|
||||
enum class SyscallOSABI;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
class Decoder final {
|
||||
@@ -24,6 +27,7 @@ public:
|
||||
SUCCESS,
|
||||
INVALID_INST,
|
||||
NOEXEC_INST,
|
||||
PARTIAL_DECODE_INST,
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
@@ -34,6 +38,7 @@ public:
|
||||
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
|
||||
DecodedBlockStatus BlockStatus;
|
||||
bool IsEntryPoint {};
|
||||
bool ForceFullSMCDetection {};
|
||||
};
|
||||
|
||||
struct DecodedBlockInformation final {
|
||||
@@ -45,7 +50,7 @@ public:
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
@@ -86,6 +91,7 @@ private:
|
||||
DecodedBlockStatus DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
bool IsBranchMonoTailcall(uint64_t NumInstructions) const;
|
||||
bool InstCanContinue() const;
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
@@ -109,6 +115,7 @@ private:
|
||||
|
||||
uint64_t ExecutableRangeBase {};
|
||||
uint64_t ExecutableRangeEnd {};
|
||||
bool ExecutableRangeWritable {};
|
||||
bool HitNonExecutableRange {};
|
||||
|
||||
const uint8_t* InstStream {};
|
||||
@@ -119,6 +126,7 @@ private:
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize {};
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
uint8_t LastEscapePrefix {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
// This is for multiblock data tracking
|
||||
@@ -147,6 +155,11 @@ private:
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_16,
|
||||
};
|
||||
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_X87_TABLE_SIZE>* X87Table;
|
||||
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -340,7 +340,7 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return sin(src);
|
||||
}
|
||||
@@ -348,7 +348,7 @@ struct OpHandlers<IR::OP_F64SIN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return cos(src);
|
||||
}
|
||||
@@ -356,7 +356,7 @@ struct OpHandlers<IR::OP_F64COS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SINCOS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
double sin, cos;
|
||||
#ifdef _WIN32
|
||||
@@ -371,7 +371,7 @@ struct OpHandlers<IR::OP_F64SINCOS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return tan(src);
|
||||
}
|
||||
@@ -379,7 +379,7 @@ struct OpHandlers<IR::OP_F64TAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
@@ -387,7 +387,7 @@ struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
@@ -395,7 +395,7 @@ struct OpHandlers<IR::OP_F64ATAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
@@ -403,7 +403,7 @@ struct OpHandlers<IR::OP_F64FPREM> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
@@ -411,7 +411,7 @@ struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
@@ -419,7 +419,7 @@ struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
@@ -445,7 +445,6 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
uint64_t Tmp = Src1.ToI64(&State.State);
|
||||
X80SoftFloat Rv;
|
||||
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
|
||||
@@ -82,24 +82,22 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle)};
|
||||
|
||||
// Double Precision Unary
|
||||
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
|
||||
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
|
||||
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_I16_F64_PTR],
|
||||
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
|
||||
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
|
||||
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
|
||||
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_I16_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
|
||||
// Double Precision Binary
|
||||
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
|
||||
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
|
||||
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
|
||||
|
||||
// SSE4.2 string instructions
|
||||
@@ -220,21 +218,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64x2_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_I16_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
|
||||
@@ -19,8 +19,8 @@ enum FallbackABI {
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_F64_F64_PTR,
|
||||
FABI_F64_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
@@ -28,7 +28,7 @@ enum FallbackABI {
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_F80x2_I16_F80_PTR,
|
||||
FABI_F64x2_I16_F64_PTR,
|
||||
FABI_F64x2_F64_PTR,
|
||||
FABI_I32_I64_I64_V128_V128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
FABI_UNKNOWN,
|
||||
|
||||
@@ -372,7 +372,7 @@ DEF_OP(CondSubNZCV) {
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
|
||||
if (Op->Cond == FEXCore::IR::COND_AL) {
|
||||
if (Op->Cond == IR::CondClass::AL) {
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
|
||||
} else {
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
|
||||
@@ -515,6 +515,12 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AndShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
|
||||
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
|
||||
@@ -582,7 +588,7 @@ DEF_OP(ShiftFlags) {
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
(void)cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
@@ -646,7 +652,7 @@ DEF_OP(ShiftFlags) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
|
||||
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
|
||||
if (PFOutput != PFTemp) {
|
||||
@@ -663,7 +669,7 @@ DEF_OP(RotateFlags) {
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
(void)cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
|
||||
@@ -695,7 +701,7 @@ DEF_OP(RotateFlags) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
@@ -761,14 +767,14 @@ DEF_OP(PDep) {
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
cbz(EmitSize, OrigMask, &Done);
|
||||
(void)cbz(EmitSize, OrigMask, &Done);
|
||||
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
and_(EmitSize, T0, T0, Mask);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
(void)Bind(&NextBit);
|
||||
sbfx(EmitSize, T1, Input, 0, 1);
|
||||
eor(EmitSize, Mask, Mask, T0);
|
||||
and_(EmitSize, T0, T1, T0);
|
||||
@@ -776,10 +782,10 @@ DEF_OP(PDep) {
|
||||
orr(EmitSize, Dest, Dest, T0);
|
||||
lsr(EmitSize, Input, Input, 1);
|
||||
and_(EmitSize, T0, Mask, T1);
|
||||
cbnz(EmitSize, T0, &NextBit);
|
||||
(void)cbnz(EmitSize, T0, &NextBit);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -815,27 +821,27 @@ DEF_OP(PExt) {
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
(void)cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
mov(EmitSize, ValueReg, Input);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
cbz(EmitSize, MaskReg, &Done);
|
||||
(void)Bind(&NextBit);
|
||||
(void)cbz(EmitSize, MaskReg, &Done);
|
||||
clz(EmitSize, BitReg, MaskReg);
|
||||
lslv(EmitSize, ValueReg, ValueReg, BitReg);
|
||||
lslv(EmitSize, MaskReg, MaskReg, BitReg);
|
||||
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
|
||||
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
|
||||
b(&NextBit);
|
||||
(void)b(&NextBit);
|
||||
|
||||
// Early exit
|
||||
Bind(&EarlyExit);
|
||||
(void)Bind(&EarlyExit);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -903,7 +909,7 @@ DEF_OP(Div) {
|
||||
eor(EmitSize, TMP1, TMP1, Upper);
|
||||
|
||||
// If the sign bit matches then the result is zero
|
||||
cbz(EmitSize, TMP1, &Only64Bit);
|
||||
(void)cbz(EmitSize, TMP1, &Only64Bit);
|
||||
|
||||
// Long divide
|
||||
{
|
||||
@@ -922,17 +928,17 @@ DEF_OP(Div) {
|
||||
mov(EmitSize, Remainder, TMP2);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
(void)b(&LongDIVRet);
|
||||
}
|
||||
|
||||
Bind(&Only64Bit);
|
||||
(void)Bind(&Only64Bit);
|
||||
// 64-Bit only
|
||||
{
|
||||
sdiv(EmitSize, Quotient, Lower, Divisor);
|
||||
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
|
||||
}
|
||||
|
||||
Bind(&LongDIVRet);
|
||||
(void)Bind(&LongDIVRet);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
|
||||
@@ -986,7 +992,7 @@ DEF_OP(UDiv) {
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
cbz(EmitSize, Upper, &Only64Bit);
|
||||
(void)cbz(EmitSize, Upper, &Only64Bit);
|
||||
|
||||
// Long divide
|
||||
{
|
||||
@@ -1005,17 +1011,17 @@ DEF_OP(UDiv) {
|
||||
mov(EmitSize, Remainder, TMP2);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
(void)b(&LongDIVRet);
|
||||
}
|
||||
|
||||
Bind(&Only64Bit);
|
||||
(void)Bind(&Only64Bit);
|
||||
// 64-Bit only
|
||||
{
|
||||
udiv(EmitSize, Quotient, Lower, Divisor);
|
||||
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
|
||||
}
|
||||
|
||||
Bind(&LongDIVRet);
|
||||
(void)Bind(&LongDIVRet);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
|
||||
@@ -1040,24 +1046,19 @@ DEF_OP(Popcount) {
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
cnt(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
|
||||
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
@@ -1189,6 +1190,19 @@ DEF_OP(Rev) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Rbit) {
|
||||
auto Op = IROp->C<IR::IROp_Rbit>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
@@ -1272,6 +1286,16 @@ DEF_OP(Sbfe) {
|
||||
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
DEF_OP(MaskGenerateFromBitWidth) {
|
||||
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
|
||||
auto BitWidth = GetReg(Op->BitWidth);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
|
||||
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
|
||||
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
|
||||
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1368,12 +1392,12 @@ DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
[[maybe_unused]] constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
@@ -33,7 +33,7 @@ void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
@@ -63,7 +63,7 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
|
||||
Bind(&Lit.Loc);
|
||||
BindOrRestart(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
@@ -77,39 +77,36 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
|
||||
const char* EntryRelocations) {
|
||||
size_t DataIndex {};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
|
||||
const auto OrigBase = GetBufferBase();
|
||||
const auto OrigSize = GetBufferSize();
|
||||
const auto OrigOffset = GetCursorOffset();
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
|
||||
for (auto& Reloc : Relocations) {
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
SetCursorOffset(Reloc.NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
@@ -117,18 +114,27 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
SetCursorOffset(Reloc.GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
|
||||
return std::move(Relocations);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -62,27 +62,27 @@ DEF_OP(CASPair) {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected0);
|
||||
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
mov(EmitSize, Dst0, Expected0);
|
||||
mov(EmitSize, Dst1, Expected1);
|
||||
|
||||
b(&LoopExpected);
|
||||
(void)b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
(void)Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst0, TMP2.R());
|
||||
mov(EmitSize, Dst1, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
(void)Bind(&LoopExpected);
|
||||
|
||||
// Restore
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
@@ -114,7 +114,7 @@ DEF_OP(CAS) {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
@@ -123,18 +123,18 @@ DEF_OP(CAS) {
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
}
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, Dst, Expected);
|
||||
b(&LoopExpected);
|
||||
(void)b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
(void)Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
(void)Bind(&LoopExpected);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -150,11 +150,11 @@ DEF_OP(AtomicXor) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -179,10 +179,10 @@ DEF_OP(AtomicSwap) {
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
stlxr(SubEmitSize, TMP4, Src, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
|
||||
}
|
||||
}
|
||||
@@ -199,11 +199,11 @@ DEF_OP(AtomicFetchAdd) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
add(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -221,11 +221,11 @@ DEF_OP(AtomicFetchSub) {
|
||||
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
sub(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -243,11 +243,11 @@ DEF_OP(AtomicFetchAnd) {
|
||||
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
and_(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -264,11 +264,11 @@ DEF_OP(AtomicFetchCLR) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
bic(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -285,11 +285,11 @@ DEF_OP(AtomicFetchOr) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
orr(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -306,11 +306,11 @@ DEF_OP(AtomicFetchXor) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -322,13 +322,26 @@ DEF_OP(AtomicFetchNeg) {
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
// Use a CAS loop to avoid needing to emulate unaligned LLSC atomics
|
||||
ldr(SubEmitSize, TMP2, MemSrc);
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
(void)Bind(&LoopTop);
|
||||
mov(EmitSize, TMP4, TMP2);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
casal(SubEmitSize, TMP2, TMP3, MemSrc);
|
||||
sub(EmitSize, TMP3, TMP2, TMP4);
|
||||
(void)cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TelemetrySetValue) {
|
||||
@@ -346,11 +359,11 @@ DEF_OP(TelemetrySetValue) {
|
||||
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
|
||||
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -141,15 +141,24 @@ DEF_OP(ExitFunction) {
|
||||
if (!Op->CallReturnBlock.IsInvalid()) {
|
||||
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
|
||||
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
|
||||
adr(TMP1, &l_CallReturn);
|
||||
(void)adr(TMP1, &l_CallReturn);
|
||||
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
|
||||
} else {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
}
|
||||
} else if (Op->Hint == IR::BranchHint::CheckTF) {
|
||||
ARMEmitter::ForwardLabel TFUnset;
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
blr(TMP2);
|
||||
(void)Bind(&TFUnset);
|
||||
}
|
||||
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
|
||||
Bind(&l_CallReturn);
|
||||
(void)Bind(&l_CallReturn);
|
||||
#ifdef _M_ARM_64EC
|
||||
}
|
||||
#endif
|
||||
@@ -161,40 +170,38 @@ DEF_OP(ExitFunction) {
|
||||
// First try to pop from the call-ret stack, otherwise follow the normal path (but ending in a ret)
|
||||
ldp<ARMEmitter::IndexType::POST>(TMP1, TMP2, REG_CALLRET_SP, 0x10);
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
}
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
|
||||
Bind(&SkipFullLookup);
|
||||
(void)Bind(&SkipFullLookup);
|
||||
if (Op->Hint == IR::BranchHint::Call) {
|
||||
ARMEmitter::ForwardLabel l_CallReturn;
|
||||
if (!Op->CallReturnBlock.IsInvalid()) {
|
||||
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
|
||||
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
|
||||
adr(TMP1, &l_CallReturn);
|
||||
(void)adr(TMP1, &l_CallReturn);
|
||||
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
|
||||
} else {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
}
|
||||
blr(TMP2);
|
||||
Bind(&l_CallReturn);
|
||||
(void)Bind(&l_CallReturn);
|
||||
} else if (Op->Hint == IR::BranchHint::Return) {
|
||||
ret(TMP2);
|
||||
} else {
|
||||
@@ -205,21 +212,20 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Target = Op->TargetBlock;
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target.ID()).first->second;
|
||||
PendingTargetLabel = JumpTarget(Op->TargetBlock);
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
auto TrueTargetLabel = JumpTarget(Op->TrueBlock);
|
||||
|
||||
if (Op->FromNZCV) {
|
||||
b(MapCC(Op->Cond), TrueTargetLabel);
|
||||
b_OrRestart(MapCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1);
|
||||
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
@@ -227,24 +233,24 @@ DEF_OP(CondJump) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
if (Op->Cond == IR::CondClass::EQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
cbz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
tbz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
|
||||
}
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
PendingTargetLabel = JumpTarget(Op->FalseBlock);
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
@@ -254,16 +260,10 @@ DEF_OP(Syscall) {
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) == FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
// Need to spill all caller saved registers still
|
||||
GPRSpillMask = CALLER_GPR_MASK;
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
@@ -297,117 +297,22 @@ DEF_OP(Syscall) {
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X8: SyscallNumber - RA INTERSECT
|
||||
// X0: Arg0 & Return
|
||||
// X1: Arg1
|
||||
// X2: Arg2
|
||||
// X3: Arg3
|
||||
// X4: Arg4 - RA INTERSECT
|
||||
// X5: Arg5 - RA INTERSECT
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
|
||||
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
|
||||
|
||||
bool Intersects {};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r1);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -438,48 +343,50 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
|
||||
auto OldCode = Op->CodeOriginal.data();
|
||||
auto Base = GetReg(Op->Header.Args[0]).X();
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
|
||||
int Offset = 0;
|
||||
ARMEmitter::ForwardLabel Fail;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
while (len >= 8) {
|
||||
ldr(ARMEmitter::XReg::x2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint64_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 8;
|
||||
idx += 8;
|
||||
}
|
||||
while (len >= 4) {
|
||||
ldr(ARMEmitter::WReg::w2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 4;
|
||||
idx += 4;
|
||||
}
|
||||
while (len >= 2) {
|
||||
ldrh(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 2;
|
||||
idx += 2;
|
||||
}
|
||||
while (len >= 1) {
|
||||
ldrb(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 1;
|
||||
idx += 1;
|
||||
}
|
||||
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
|
||||
while (len >= Size) {
|
||||
LoadData();
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
len -= Size;
|
||||
Offset += Size;
|
||||
}
|
||||
};
|
||||
|
||||
EmitCheck(8, [&]() {
|
||||
ldr(TMP1, Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(4, [&]() {
|
||||
ldr(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(2, [&]() {
|
||||
ldrh(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(1, [&]() {
|
||||
ldrb(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
|
||||
@@ -423,11 +423,11 @@ DEF_OP(Vector_FToI) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
} else {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
@@ -449,21 +449,21 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
|
||||
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
|
||||
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
|
||||
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
|
||||
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
|
||||
case IR::RoundMode::Nearest: ROUNDING_FN(frintn); break;
|
||||
case IR::RoundMode::NegInfinity: ROUNDING_FN(frintm); break;
|
||||
case IR::RoundMode::PosInfinity: ROUNDING_FN(frintp); break;
|
||||
case IR::RoundMode::TowardsZero: ROUNDING_FN(frintz); break;
|
||||
case IR::RoundMode::Host: ROUNDING_FN(frinti); break;
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -539,11 +539,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
// Then convert to integers using fcvtzs.
|
||||
auto CVTReg = Dst.Z();
|
||||
switch (Round) {
|
||||
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
|
||||
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Z(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
|
||||
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -567,11 +567,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
switch (Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
// Now narrow from f64 to f32.
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
};
|
||||
|
||||
struct DebugDataGuestOpcode {
|
||||
uint64_t GuestEntryOffset;
|
||||
ptrdiff_t HostEntryOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Contains debug data for a block of code for later debugger analysis
|
||||
*
|
||||
* Needs to remain around for as long as the code could be executed at least
|
||||
*/
|
||||
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
fextl::vector<DebugDataSubblock> Subblocks;
|
||||
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
|
||||
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
|
||||
};
|
||||
} // namespace FEXCore::Core
|
||||
@@ -16,7 +16,7 @@ DEF_OP(VAESImc) {
|
||||
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -41,7 +41,7 @@ DEF_OP(VAESEnc) {
|
||||
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -64,7 +64,7 @@ DEF_OP(VAESEncLast) {
|
||||
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -89,7 +89,7 @@ DEF_OP(VAESDec) {
|
||||
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -322,7 +322,7 @@ DEF_OP(VSha256U1) {
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
|
||||
@@ -11,15 +11,12 @@ desc: Main glue logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/JIT/DebugData.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
@@ -30,15 +27,16 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
namespace {
|
||||
struct DivRem {
|
||||
@@ -222,7 +220,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
case FABI_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -239,7 +237,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64x2_I16_F64_PTR: {
|
||||
case FABI_F64x2_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -264,7 +262,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
FillF64x2Result(DstLo, DstHi);
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
case FABI_F64_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -538,8 +536,9 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
} else {
|
||||
{
|
||||
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
auto lk_inval =
|
||||
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(Thread, GuestRip);
|
||||
}
|
||||
if (!HostCode) {
|
||||
// Hold a reference to the code buffer, to avoid linking unmapped code if compilation triggers a recreation.
|
||||
@@ -564,7 +563,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// Lock here is necessary to prevent simultaneous linking and delinking
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
|
||||
// For non-calls, this would extend into the block's code, however that's fine as an out-of-range adr would never
|
||||
// be generated avoiding any false positives.
|
||||
@@ -577,14 +576,17 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
|
||||
if (KnownCallMarkerInst == ExpectedKnownCallMarkerInst) {
|
||||
BranchEmit.bl(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, true);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Frame, Record, true); }, lk);
|
||||
} else {
|
||||
BranchEmit.b(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, false);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, false);
|
||||
},
|
||||
lk);
|
||||
}
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
@@ -603,7 +605,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(LdrInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker, lk);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
@@ -624,10 +626,10 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPR, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
@@ -639,6 +641,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Common.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
@@ -667,15 +670,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
} else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -687,13 +681,13 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -739,48 +733,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
void Arm64JITCore::EmitTFCheck() {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
(void)tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
(void)Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
@@ -793,19 +787,21 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
(void)Bind(&l_NoSuspend);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
|
||||
// Get the address of the JITCodeHeader and store in to the core state.
|
||||
// Two instruction cost, each 1 cycle.
|
||||
adr(TMP1, &HeaderLabel);
|
||||
adr_OrRestart(TMP1, &HeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
EmitInterruptChecks(CheckTF);
|
||||
if (CheckTF) {
|
||||
EmitTFCheck();
|
||||
}
|
||||
|
||||
if (SpillSlots) {
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
@@ -817,20 +813,34 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
RequiresFarARM64Jumps = false;
|
||||
|
||||
// Prepare restart via long jump in case branch encoding fails.
|
||||
// This uses UncheckedLongJump since we don't implement std::longjmp in WoA setups
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::UncheckedLongJump::SetJump(RestartControl.RestartJump))) {
|
||||
case RestartOptions::Control::Incoming:
|
||||
// Nothing
|
||||
break;
|
||||
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
|
||||
default: ERROR_AND_DIE_FMT("Unhandled Arm64 restart condition!");
|
||||
}
|
||||
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
JumpTargets.clear();
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
|
||||
|
||||
CodeData.EntryPoints.clear();
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
@@ -845,7 +855,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
Bind(&JITCodeHeaderLabel);
|
||||
(void)Bind(&JITCodeHeaderLabel);
|
||||
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
|
||||
CursorIncrement(sizeof(JITCodeHeader));
|
||||
|
||||
@@ -886,11 +896,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t*>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
const auto Target = &JumpTargets[BlockIROp->ID];
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
|
||||
b(PendingTargetLabel);
|
||||
if (PendingTargetLabel && PendingTargetLabel != Target) {
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b_OrRestart(PendingTargetLabel);
|
||||
PendingTargetLabel = nullptr;
|
||||
}
|
||||
|
||||
@@ -900,14 +913,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
const auto IsReturnTarget = CallReturnTargets.try_emplace(Node).first;
|
||||
if (PendingTargetLabel) {
|
||||
// If there is a fallthrough branch to this block, skip over the entrypoint code.
|
||||
b(&IsTarget->second);
|
||||
b_OrRestart(Target);
|
||||
} else if (PendingCallReturnTargetLabel && PendingCallReturnTargetLabel != &IsReturnTarget->second) {
|
||||
// If we just emitted a call, but the block we're now emitting is not the return block so don't fallthrough.
|
||||
b(PendingCallReturnTargetLabel);
|
||||
b_OrRestart(PendingCallReturnTargetLabel);
|
||||
}
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
|
||||
Bind(&IsReturnTarget->second);
|
||||
BindOrRestart(&IsReturnTarget->second);
|
||||
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
|
||||
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
|
||||
|
||||
@@ -916,18 +929,16 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
if (PendingCallReturnTargetLabel) {
|
||||
// If there is still a pending call return target, then the block we're emitting is not the return block so don't fallthrough.
|
||||
b(PendingCallReturnTargetLabel);
|
||||
b_OrRestart(PendingCallReturnTargetLabel);
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
Bind(&IsTarget->second);
|
||||
BindOrRestart(Target);
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
|
||||
|
||||
@@ -945,7 +956,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel) {
|
||||
b(PendingTargetLabel);
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b_OrRestart(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
@@ -956,21 +970,21 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
ARMEmitter::ForwardLabel l_DoLink;
|
||||
uint64_t ThunkAddress = GetCursorAddress<uint64_t>();
|
||||
Bind(&PendingJumpThunk.Label);
|
||||
b(&l_DoLink);
|
||||
BindOrRestart(&PendingJumpThunk.Label);
|
||||
b_OrRestart(&l_DoLink);
|
||||
br(TMP1);
|
||||
Bind(&l_DoLink);
|
||||
BindOrRestart(&l_DoLink);
|
||||
ldr(TMP1, &l_ExitLink);
|
||||
blr(TMP1);
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
Bind(&l_ExitLink);
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
Bind(&l_ExitLink);
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
|
||||
// CodeSize not including the header or tail data.
|
||||
@@ -1054,7 +1068,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
|
||||
@@ -10,30 +10,39 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct ExitFunctionLinkData;
|
||||
}
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
@@ -52,14 +61,25 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
const bool HostSupportsAVX256 {};
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
struct RestartOptions {
|
||||
FEXCore::UncheckedLongJump::JumpBuf RestartJump;
|
||||
enum class Control : uint64_t {
|
||||
Incoming = 0,
|
||||
EnableFarARM64Jumps = 1,
|
||||
};
|
||||
};
|
||||
|
||||
// FEXCore makes assumptions in the JIT about certain conditions being true.
|
||||
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
|
||||
RestartOptions RestartControl {};
|
||||
bool RequiresFarARM64Jumps {};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
@@ -67,7 +87,13 @@ private:
|
||||
uint64_t Entry {};
|
||||
CPUBackend::CompiledCode CodeData {};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
fextl::vector<ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* JumpTarget(IR::OrderedNodeWrapper Node) {
|
||||
auto Block = IR->GetOp<IR::IROp_CodeBlock>(Node);
|
||||
return &JumpTargets[Block->ID];
|
||||
}
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> CallReturnTargets;
|
||||
|
||||
struct PendingJumpThunk {
|
||||
@@ -83,11 +109,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::GPRFixed || RegClass == IR::RegClass::GPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
} else if (RegClass == IR::RegClass::GPR) {
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -106,11 +134,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::FPRFixed || RegClass == IR::RegClass::FPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::FPRFixed) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
} else if (RegClass == IR::RegClass::FPR) {
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -128,8 +158,8 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
|
||||
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
|
||||
static IR::RegClass GetRegClass(IR::Ref Node) {
|
||||
return IR::PhysicalRegister(Node).AsRegClass();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -146,7 +176,7 @@ private:
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
static ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
@@ -154,18 +184,23 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
static ARMEmitter::Size ConvertSize(IR::OpSize Size) {
|
||||
return Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
@@ -177,105 +212,105 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize16(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize8(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
static ARMEmitter::Condition MapCC(IR::CondClass Cond) {
|
||||
switch (Cond) {
|
||||
case IR::CondClass::EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case IR::CondClass::NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case IR::CondClass::SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case IR::CondClass::ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case IR::CondClass::UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case IR::CondClass::ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case IR::CondClass::FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::FU:
|
||||
case IR::CondClass::VS: return ARMEmitter::Condition::CC_VS;
|
||||
case IR::CondClass::FNU:
|
||||
case IR::CondClass::VC: return ARMEmitter::Condition::CC_VC;
|
||||
case IR::CondClass::MI: return ARMEmitter::Condition::CC_MI;
|
||||
case IR::CondClass::PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
static bool IsFPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::FPR || Class == IR::RegClass::FPRFixed;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
static bool IsGPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::GPR || Class == IR::RegClass::GPRFixed;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::Ref Node) {
|
||||
static bool IsGPR(IR::Ref Node) {
|
||||
return IsGPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::Ref Node) {
|
||||
static bool IsFPR(IR::Ref Node) {
|
||||
return IsFPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
static bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
static bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -308,14 +343,179 @@ private:
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
|
||||
auto& Thunk = PendingJumpThunks.back();
|
||||
Bind(&Thunk.Label);
|
||||
BindOrRestart(&Thunk.Label);
|
||||
if (Call) {
|
||||
bl(&Thunk.Label);
|
||||
bl_OrRestart(&Thunk.Label);
|
||||
} else {
|
||||
b(&Thunk.Label);
|
||||
b_OrRestart(&Thunk.Label);
|
||||
}
|
||||
}
|
||||
|
||||
// Restart helpers
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void bl_OrRestart(T* Label) {
|
||||
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void b_OrRestart(T* Label) {
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)b(InvertCondition(Cond), &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (b(Cond, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)cbnz(s, rt, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (cbz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)cbz(s, rt, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (cbnz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)tbnz(rt, Bit, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tbz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)tbz(rt, Bit, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tbnz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADR currently unsupported!");
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADRP currently unsupported!");
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void BindOrRestart(T* Label) {
|
||||
if (Bind(Label)) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (RequiresFarARM64Jumps) {
|
||||
// This should have been caught before this point.
|
||||
ERROR_AND_DIE_FMT("Unhandled long bind");
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
@@ -374,7 +574,9 @@ private:
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
|
||||
|
||||
/** @} */
|
||||
|
||||
@@ -397,23 +599,16 @@ private:
|
||||
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize);
|
||||
|
||||
void EmitInterruptChecks(bool CheckTF);
|
||||
void EmitTFCheck();
|
||||
|
||||
void EmitSuspendInterruptCheck();
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -52,7 +52,7 @@ DEF_OP(LoadContext) {
|
||||
DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -78,7 +78,7 @@ DEF_OP(StoreContext) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -110,7 +110,7 @@ DEF_OP(StoreContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Src1 = GetZeroableReg(Op->Value1);
|
||||
auto Src2 = GetZeroableReg(Op->Value2);
|
||||
|
||||
@@ -135,12 +135,12 @@ DEF_OP(StoreContextPair) {
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
} else if (Op->Class == IR::RegClass::FPR) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
@@ -175,13 +175,14 @@ DEF_OP(LoadAF) {
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass) {
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
|
||||
} else if (Reg.Class == IR::FPRFixedClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
} else if (RegClass == IR::RegClass::FPRFixed) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
const auto guest = GetVReg(Reg);
|
||||
@@ -193,7 +194,7 @@ DEF_OP(StoreRegister) {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", RegClass);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,7 +226,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
@@ -288,7 +289,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Value = GetReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -348,12 +349,31 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FormContextAddress) {
|
||||
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
|
||||
const auto Index = GetReg(Op->Index);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -394,7 +414,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -433,7 +453,7 @@ DEF_OP(SpillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -442,7 +462,7 @@ DEF_OP(FillRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -483,7 +503,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -522,7 +542,7 @@ DEF_OP(FillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -559,14 +579,14 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
case IR::MemOffsetType::UXTW:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
case IR::MemOffsetType::SXTW:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -593,20 +613,20 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
case IR::MemOffsetType::UXTW:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
case IR::MemOffsetType::SXTW:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType); break;
|
||||
}
|
||||
}
|
||||
return Tmp;
|
||||
@@ -657,7 +677,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
// Note that we do nothing with the offset type and offset scale,
|
||||
// since SVE loads and stores don't have the ability to perform an
|
||||
// optional extension or shift as part of their behavior.
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
LOGMAN_THROW_A_FMT(OffsetType == IR::MemOffsetType::SXTX, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
@@ -670,7 +690,7 @@ DEF_OP(LoadMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -704,7 +724,7 @@ DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -732,17 +752,17 @@ DEF_OP(LoadMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
@@ -760,7 +780,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -775,7 +795,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -892,7 +912,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
@@ -903,7 +923,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
@@ -993,7 +1013,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
@@ -1004,7 +1024,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
@@ -1020,7 +1040,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister IncomingDst, std::optional<ARMEmitter::Register> BaseAddr,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
@@ -1082,7 +1102,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
PerformMove(ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// Skip if the mask's sign bit isn't set
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
|
||||
// Extract Index Element
|
||||
if ((IndexElement * IR::OpSizeToSize(VectorIndexSize)) >= 16) {
|
||||
@@ -1096,17 +1116,17 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
// Calculate memory position for this gather load
|
||||
if (BaseAddr.has_value()) {
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
} else {
|
||||
///< In this case we have no base address, All addresses come from the vector register itself
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
// Sign extend and shift in to the 64-bit register
|
||||
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
sbfiz(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
} else {
|
||||
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
lsl(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1120,7 +1140,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ElementSize); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
}
|
||||
|
||||
if (NeedsDestTmp) {
|
||||
@@ -1164,7 +1184,8 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) && VectorIndexSize == IROp->ElementSize;
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) &&
|
||||
VectorIndexSize == IROp->ElementSize && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
|
||||
@@ -1222,7 +1243,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
|
||||
Emulate128BitGather(IROp->Size, IROp->ElementSize, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale);
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale, Op->AddrSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1247,7 +1268,9 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
|
||||
const bool SupportsSVELoad = HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4) && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
|
||||
if (OffsetScale != 1) {
|
||||
ModType = ARMEmitter::SVEModType::MOD_LSL;
|
||||
@@ -1301,7 +1324,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
}
|
||||
} else {
|
||||
Emulate128BitGather(IR::OpSize::i128Bit, IR::OpSize::i32Bit, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale);
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale, Op->AddrSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1602,7 +1625,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
@@ -1713,7 +1736,7 @@ DEF_OP(StoreMemPair) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
@@ -1740,17 +1763,17 @@ DEF_OP(StoreMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
@@ -1767,7 +1790,7 @@ DEF_OP(StoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -1851,7 +1874,7 @@ DEF_OP(MemSet) {
|
||||
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
(void)tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
|
||||
@@ -1899,7 +1922,7 @@ DEF_OP(MemSet) {
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
@@ -1916,50 +1939,50 @@ DEF_OP(MemSet) {
|
||||
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
|
||||
// single copy loop if size < 64.
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
|
||||
// Fill VTMP2 with the set pattern
|
||||
dup(SubRegSize, VTMP2.Q(), Value);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
|
||||
Bind(&AgainInternal256);
|
||||
(void)Bind(&AgainInternal256);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal256);
|
||||
(void)tbz(TMP1, 63, &AgainInternal256);
|
||||
|
||||
Bind(&AgainInternal256Exit);
|
||||
(void)Bind(&AgainInternal256Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
Bind(&AgainInternal128);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal128);
|
||||
(void)tbz(TMP1, 63, &AgainInternal128);
|
||||
|
||||
Bind(&AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&AgainInternal);
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemStoreTSO(Value, OpSize, SizeDirection);
|
||||
} else {
|
||||
MemStore(Value, OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
@@ -1989,12 +2012,12 @@ DEF_OP(MemSet) {
|
||||
EmitMemset(Direction);
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
(void)b(&Done);
|
||||
(void)Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
@@ -2044,7 +2067,7 @@ DEF_OP(MemCpy) {
|
||||
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
(void)tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
@@ -2141,7 +2164,7 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
ARMEmitter::ForwardLabel AbsPos {};
|
||||
@@ -2151,11 +2174,11 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::BackwardLabel AgainInternal256 {};
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
|
||||
tbz(TMP4, 63, &AbsPos);
|
||||
(void)tbz(TMP4, 63, &AbsPos);
|
||||
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
|
||||
Bind(&AbsPos);
|
||||
(void)Bind(&AbsPos);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
|
||||
tbnz(TMP4, 63, &AgainInternal);
|
||||
(void)tbnz(TMP4, 63, &AgainInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
@@ -2167,30 +2190,30 @@ DEF_OP(MemCpy) {
|
||||
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
|
||||
// single copy loop if size < 64.
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
|
||||
Bind(&AgainInternal256);
|
||||
(void)Bind(&AgainInternal256);
|
||||
MemCpy(32, 32 * Direction);
|
||||
MemCpy(32, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal256);
|
||||
(void)tbz(TMP1, 63, &AgainInternal256);
|
||||
|
||||
Bind(&AgainInternal256Exit);
|
||||
(void)Bind(&AgainInternal256Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
Bind(&AgainInternal128);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128);
|
||||
MemCpy(32, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal128);
|
||||
(void)tbz(TMP1, 63, &AgainInternal128);
|
||||
|
||||
Bind(&AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
@@ -2198,16 +2221,16 @@ DEF_OP(MemCpy) {
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&AgainInternal);
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemCpyTSO(OpSize, SizeDirection);
|
||||
} else {
|
||||
MemCpy(OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
// Needs to use temporaries just in case of overwrite
|
||||
mov(TMP1, MemRegDest.X());
|
||||
@@ -2265,186 +2288,15 @@ DEF_OP(MemCpy) {
|
||||
for (int32_t Direction : {1, -1}) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
(void)b(&Done);
|
||||
(void)Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: ldapurh(Dst, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: ldapur(Dst.W(), MemReg, Offset); break;
|
||||
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: ldaprh(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i32Bit: ldapr(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
|
||||
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
|
||||
case IR::OpSize::i32Bit: ldar(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case IR::OpSize::i128Bit:
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
|
||||
case IR::OpSize::i64Bit: stlur(Src.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: stlr(Src.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i128Bit: {
|
||||
// Move vector to GPRs
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP2, Src, 1);
|
||||
ARMEmitter::BackwardLabel B;
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::SY);
|
||||
@@ -2586,7 +2438,7 @@ DEF_OP(VStoreNonTemporalPair) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreNonTemporalPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
[[maybe_unused]] const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "This IR operation only operates at 128-bit wide");
|
||||
|
||||
const auto ValueLow = GetVReg(Op->ValueLow);
|
||||
|
||||
@@ -10,10 +10,12 @@ $end_info$
|
||||
#endif
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/DebugData.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -46,10 +48,10 @@ DEF_OP(GuestOpcode) {
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::Fence_Inst.Val: isb(); break;
|
||||
case IR::FenceType::Load: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::FenceType::LoadStore: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::FenceType::Store: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::FenceType::Inst: isb(); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
@@ -106,10 +108,10 @@ DEF_OP(GetRoundingMode) {
|
||||
// zero. Just swapping 01 and 10. That's a bitfield reverse. Round mode is in
|
||||
// bottom two bits. After reversing as a 32-bit operation, it'll be in [31:30]
|
||||
// and ripe for reinsertion back at 0.
|
||||
static_assert(IR::ROUND_MODE_NEAREST == 0);
|
||||
static_assert(IR::ROUND_MODE_NEGATIVE_INFINITY == 1);
|
||||
static_assert(IR::ROUND_MODE_POSITIVE_INFINITY == 2);
|
||||
static_assert(IR::ROUND_MODE_TOWARDS_ZERO == 3);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::Nearest) == 0);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::NegInfinity) == 1);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::PosInfinity) == 2);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::TowardsZero) == 3);
|
||||
|
||||
rbit(ARMEmitter::Size::i32Bit, TMP1, Dst);
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP1, 30, 2);
|
||||
@@ -286,4 +288,43 @@ DEF_OP(Yield) {
|
||||
yield();
|
||||
}
|
||||
|
||||
DEF_OP(MonoBackpatcherWrite) {
|
||||
auto Op = IROp->C<IR::IROp_MonoBackpatcherWrite>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Addr));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4, GetReg(Op->Value));
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, IR::OpSizeToSize(Op->Size));
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, TMP4);
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.MonoBackpatcherWrite));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, uint8_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -18,18 +18,4 @@ DEF_OP(RMWHandle) {
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A), B = GetReg(Op->B);
|
||||
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
mov(ARMEmitter::Size::i64Bit, A, B);
|
||||
mov(ARMEmitter::Size::i64Bit, B, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
File renamed without changes.
@@ -41,6 +41,7 @@ namespace FEXCore::CPU {
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit; \
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
@@ -49,8 +50,10 @@ namespace FEXCore::CPU {
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
} else if (Is128Bit) { \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} else { \
|
||||
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
|
||||
} \
|
||||
}
|
||||
|
||||
@@ -744,11 +747,11 @@ DEF_OP(VFToIScalarInsert) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
switch (RoundMode) {
|
||||
case IR::Round_Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Negative_Infinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Positive_Infinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Towards_Zero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1352,7 +1355,7 @@ DEF_OP(VFMin) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1425,7 +1428,7 @@ DEF_OP(VFMax) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore {
|
||||
GuestToHostMap::GuestToHostMap()
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
|
||||
: BlockLinks_mbr {"FEXMem_BlockLinks"} {
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
@@ -24,7 +24,7 @@ GuestToHostMap::GuestToHostMap()
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE + MAX_L1_SIZE;
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -39,6 +39,10 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
@@ -46,14 +50,21 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8;
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup_L1", reinterpret_cast<void*>(L1Pointer), MAX_L1_SIZE);
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
|
||||
if (DynamicL1Cache()) {
|
||||
// Start at minimum size when dynamic.
|
||||
L1PointerMask = MIN_L1_ENTRIES - 1;
|
||||
} else {
|
||||
// Start at maximum instead.
|
||||
L1PointerMask = MAX_L1_ENTRIES - 1;
|
||||
}
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
@@ -64,31 +75,27 @@ LookupCache::~LookupCache() {
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void LookupCache::ClearL2Cache(const FEXCore::LookupCacheReadLockToken& lk) {
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
void LookupCache::ClearCache(const LookupCacheWriteLockToken& lk) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
void GuestToHostMap::ClearCache(const LookupCacheWriteLockToken&) {
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -2,6 +2,9 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include "Utils/WritePriorityMutex.h"
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
@@ -9,23 +12,40 @@
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
struct LookupCacheWriteLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheWriteLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::lock_guard<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct LookupCacheReadLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheReadLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::shared_lock<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct GuestToHostMap {
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
struct LockToken {
|
||||
std::lock_guard<std::recursive_mutex> Lock;
|
||||
};
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex Lock {};
|
||||
|
||||
[[nodiscard]]
|
||||
LockToken AcquireLock() {
|
||||
return LockToken {std::lock_guard {WriteLock}};
|
||||
LookupCacheWriteLockToken AcquireWriteLock() {
|
||||
return LookupCacheWriteLockToken {Lock};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
LookupCacheReadLockToken AcquireReadLock() {
|
||||
return LookupCacheReadLockToken {Lock};
|
||||
}
|
||||
|
||||
struct BlockLinkTag {
|
||||
@@ -49,7 +69,7 @@ struct GuestToHostMap {
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
@@ -61,7 +81,7 @@ struct GuestToHostMap {
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
@@ -70,7 +90,7 @@ struct GuestToHostMap {
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
if (HostCode == BlockList.end()) {
|
||||
return std::nullopt;
|
||||
@@ -78,7 +98,7 @@ struct GuestToHostMap {
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LookupCacheWriteLockToken&) {
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
@@ -91,11 +111,11 @@ struct GuestToHostMap {
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) {
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LockToken&) {
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) {
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -107,7 +127,7 @@ struct GuestToHostMap {
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ClearCache(const LockToken&);
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
};
|
||||
|
||||
class LookupCache {
|
||||
@@ -122,69 +142,134 @@ public:
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) {
|
||||
ClearThreadLocalCaches(lk);
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
uintptr_t FindBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
auto lk = Shared->AcquireLock();
|
||||
uintptr_t HostPtr {};
|
||||
{
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheReadLockTime : nullptr);
|
||||
auto lk = Shared->AcquireReadLock();
|
||||
LockTime.reset();
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
if (!DisableL2Cache()) {
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
HostPtr = L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!HostPtr) {
|
||||
// Try L3
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value(), lk);
|
||||
HostPtr = HostCode.value();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
if (HostPtr && DynamicL1Cache()) {
|
||||
UpdateDynamicL1Stats(Thread);
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
|
||||
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// If host pointer was found in L2 or L3, then add it to the counter.
|
||||
// Keeping track not L1 misses, but specifically L2/L3 hits.
|
||||
++L2L3CacheHits;
|
||||
|
||||
const auto CurrentTime = std::chrono::system_clock::now();
|
||||
const auto Period = CurrentTime - LastPeriod;
|
||||
if (Period >= SamplePeriod) {
|
||||
// If larger than the sample period then check if we need to increase L1 cache size.
|
||||
const double AveragePerSecond = static_cast<double>(L2L3CacheHits) /
|
||||
static_cast<double>(std::chrono::duration_cast<std::chrono::milliseconds>(Period).count()) * 1000.0;
|
||||
|
||||
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries < MAX_L1_ENTRIES) {
|
||||
CurrentL1Entries <<= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries > MIN_L1_ENTRIES) {
|
||||
CurrentL1Entries >>= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Madvise the entries that we are dropping. Gives the memory back to the OS.
|
||||
LookupCacheEntry* FirstZeroL1Entry = &reinterpret_cast<LookupCacheEntry*>(L1Pointer)[CurrentL1Entries];
|
||||
size_t ZeroMemorySize = (MAX_L1_ENTRIES - CurrentL1Entries) * sizeof(LookupCacheEntry);
|
||||
FEXCore::Allocator::VirtualDontNeed(FirstZeroL1Entry, ZeroMemorySize, false);
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
}
|
||||
|
||||
// Update Last period to start again.
|
||||
LastPeriod = CurrentTime;
|
||||
L2L3CacheHits = 0;
|
||||
}
|
||||
}
|
||||
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* HostCode) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
@@ -192,13 +277,11 @@ public:
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LookupCacheWriteLockToken& lk) {
|
||||
bool ErasedAny = Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = 0;
|
||||
ErasedAny = true;
|
||||
@@ -207,37 +290,42 @@ public:
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
if (!DisableL2Cache()) {
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return ErasedAny;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return ErasedAny;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) {
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
void ClearL2Cache(const LookupCacheReadLockToken&);
|
||||
void ClearThreadLocalCaches(const LookupCacheWriteLockToken&);
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
}
|
||||
uintptr_t GetScaledL1PointerMask() const {
|
||||
return L1PointerMask << FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry));
|
||||
}
|
||||
uintptr_t GetPagePointer() const {
|
||||
return PagePointer;
|
||||
}
|
||||
@@ -245,9 +333,6 @@ public:
|
||||
return VirtualMemSize;
|
||||
}
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
@@ -255,45 +340,48 @@ public:
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
auto AcquireLock() {
|
||||
return Shared->AcquireLock();
|
||||
auto AcquireWriteLock() {
|
||||
return Shared->AcquireWriteLock();
|
||||
}
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode, const LookupCacheReadLockToken& lk) {
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = HostCode;
|
||||
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
if (!DisableL2Cache()) {
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache();
|
||||
CacheBlockMapping(Address, HostCode);
|
||||
return;
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache(lk);
|
||||
CacheBlockMapping(Address, HostCode, lk);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = HostCode;
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = HostCode;
|
||||
}
|
||||
|
||||
uintptr_t AllocateBackingForPage() {
|
||||
@@ -313,16 +401,32 @@ private:
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
uintptr_t L1PointerMask;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
// Start with 8k entries in L1 to give 128KB of L1 cache to each thread.
|
||||
// Max out at 1 million entries to give each thread 16MB of L1 cache maximum.
|
||||
constexpr static size_t MIN_L1_ENTRIES = 8 * 1024; // Must be a power of 2
|
||||
constexpr static size_t MAX_L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t SIZE_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t MAX_L1_SIZE = MAX_L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
FEXCore::Context::ContextImpl* ctx;
|
||||
uint64_t VirtualMemSize {};
|
||||
|
||||
size_t CurrentL1Entries = MIN_L1_ENTRIES;
|
||||
uint64_t L2L3CacheHits {};
|
||||
std::chrono::time_point<std::chrono::system_clock> LastPeriod {};
|
||||
constexpr static std::chrono::seconds SamplePeriod {1};
|
||||
FEX_CONFIG_OPT(DynamicL1CacheIncreaseCountHeuristic, DYNAMICL1CACHEINCREASECOUNTHEURISTIC);
|
||||
FEX_CONFIG_OPT(DynamicL1CacheDecreaseCountHeuristic, DYNAMICL1CACHEDECREASECOUNTHEURISTIC);
|
||||
|
||||
FEX_CONFIG_OPT(DynamicL1Cache, DYNAMICL1CACHE);
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -1,86 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct FEX_PACKED CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie {};
|
||||
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock {};
|
||||
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
unsigned ABILocalFlags : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
unsigned ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
unsigned Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
unsigned x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 19;
|
||||
|
||||
bool operator==(const CodeObjectSerializationConfig& other) const {
|
||||
return Cookie == other.Cookie && MaxInstPerBlock == other.MaxInstPerBlock && Arch == other.Arch && MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled && TSOEnabled == other.TSOEnabled && ABILocalFlags == other.ABILocalFlags &&
|
||||
ParanoidTSO == other.ParanoidTSO && Is64BitMode == other.Is64BitMode && SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(const CodeObjectSerializationConfig& other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash {};
|
||||
Hash <<= 32;
|
||||
Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1;
|
||||
Hash |= other.Arch;
|
||||
Hash <<= 1;
|
||||
Hash |= other.MultiBlock;
|
||||
Hash <<= 1;
|
||||
Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1;
|
||||
Hash |= other.TSOEnabled;
|
||||
Hash <<= 1;
|
||||
Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1;
|
||||
Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1;
|
||||
Hash |= other.Is64BitMode;
|
||||
Hash <<= 2;
|
||||
Hash |= other.SMCChecks;
|
||||
Hash <<= 1;
|
||||
Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is "
|
||||
"generated!");
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,121 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
|
||||
#ifndef _WIN32
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = fextl::make_unique<CodeRegionEntry>(Base, Size, Offset, filename, NamedRegionHandler->DefaultCodeHeader(Base, Offset));
|
||||
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
} else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
#ifndef _WIN32
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
}
|
||||
} else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,71 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename,
|
||||
const fextl::string& filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion*>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion*>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,85 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void* Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService* This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler {&NamedRegionHandler, this}
|
||||
, NamedRegionHandler {ctx} {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
WorkerThreadShuttingDown = true;
|
||||
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
const CodeObjectFileSection* CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto& it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,457 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
#include "Interface/Core/ObjectCache/CodeObjectSerializationConfig.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/queue.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// XXX: Does this need to be signal safe?
|
||||
using CodeSerializationMutex = std::shared_mutex;
|
||||
struct CodeSerializationData {};
|
||||
|
||||
struct CodeObjectFileSection {
|
||||
bool Serialized;
|
||||
bool Invalid;
|
||||
const CodeSerializationData* Data;
|
||||
const char* HostCode;
|
||||
uint64_t NumRelocations;
|
||||
const char* Relocations;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is the file header that lives at the start of an object cache file
|
||||
*
|
||||
* This header is updated from multiple processes!
|
||||
* Care must be taken to use OS locks when updating the file backing including this header
|
||||
*/
|
||||
struct CodeObjectSerializationHeader {
|
||||
// The configuration that this file has
|
||||
CodeObjectSerializationConfig Config;
|
||||
// The original RIP that this object section was mapped at
|
||||
uint64_t OriginalBase {};
|
||||
// The original offset in to the file that this object section was loaded from
|
||||
uint64_t OriginalOffset {};
|
||||
// Total amount of code that should be in this file
|
||||
uint64_t TotalCodeSize {};
|
||||
// Used to reserve the TSL map
|
||||
uint64_t NumCodeEntries {};
|
||||
// The number of relocations that point to this section
|
||||
uint64_t NumRelocationsTo {};
|
||||
// Total relocations in this file
|
||||
uint64_t TotalRelocationsCount {};
|
||||
};
|
||||
|
||||
struct CodeRegionEntry {
|
||||
/**
|
||||
* @name Threaded initialization objects for the initial object creation
|
||||
* @{ */
|
||||
// Base address in memory where the code region is at
|
||||
uint64_t Base {};
|
||||
|
||||
// Size of this code entry
|
||||
uint64_t Size {};
|
||||
|
||||
// The offset inside the file that is mapped to Base
|
||||
uint64_t Offset {};
|
||||
|
||||
// Filename of the object
|
||||
fextl::string Filename {};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader {};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
fextl::string ObjectEntrySourceFilename {};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
bool StillSerializing {true};
|
||||
|
||||
// Long lived FD for serialization if we have multiple jobs to serialize
|
||||
// Bursts of code entries are common and this reduces file lock overhead
|
||||
//
|
||||
// Especially useful over network mounts where file locks are very slow
|
||||
int CurrentSerializedFD {-1};
|
||||
|
||||
/**
|
||||
* @name Objects required to sync objects between threads
|
||||
* @{ */
|
||||
// Refcount for the number of outstanding code entries waiting to be written for this object section
|
||||
CodeSerializationMutex ObjectJobRefCountMutex;
|
||||
|
||||
// Refcount for outstanding named object region entry loading itself
|
||||
// Will block JIT code cache look up when this has a unique_lock held
|
||||
CodeSerializationMutex NamedJobRefCountMutex;
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Object Entry data management
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @name This is the raw file data that we loaded from the code region entry file
|
||||
* @{ */
|
||||
char* CodeData {};
|
||||
size_t FileSize {};
|
||||
|
||||
fextl::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap {};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
CodeRegionEntry() = default;
|
||||
|
||||
// Initializer specifically for threaded loading
|
||||
CodeRegionEntry(uint64_t Base, uint64_t Size, uint64_t Offset, const fextl::string& Filename, const CodeObjectSerializationHeader& DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
, Offset {Offset}
|
||||
, Filename {Filename}
|
||||
, EntryHeader {DefaultHeader} {}
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
|
||||
class AsyncJobHandler final {
|
||||
public:
|
||||
/**
|
||||
* @brief Structure containing all the data required to async serialize code objects
|
||||
*/
|
||||
struct SerializationJobData {
|
||||
uint64_t GuestRIP; ///< The RIP for the guest
|
||||
// XXX: Support multiblock
|
||||
uint64_t GuestCodeLength; ///< The Guest's code length
|
||||
uint64_t GuestCodeHash; ///< Hash of the guest code
|
||||
|
||||
void* HostCodeBegin; ///< Host JIT code starting memory address
|
||||
size_t HostCodeLength; ///< Host JIT code length
|
||||
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
|
||||
|
||||
// This is the thread specific ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
|
||||
// This way it will wait until the async job handler is complete with it.
|
||||
CodeSerializationMutex* ThreadJobRefCount;
|
||||
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
* @{ */
|
||||
// This is the code region's ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
|
||||
CodeSerializationMutex* ObjectJobRefCountMutexPtr;
|
||||
|
||||
// This is the code region iterator to reduce the number of map lookups
|
||||
// This will remain valid while jobs are outstanding for this region
|
||||
CodeRegionMapType::iterator CodeRegionIterator;
|
||||
/** @} */
|
||||
};
|
||||
|
||||
AsyncJobHandler(NamedRegionObjectHandler* NamedRegionHandler, CodeObjectSerializeService* CodeObjectCacheService)
|
||||
: NamedRegionHandler {NamedRegionHandler}
|
||||
, CodeObjectCacheService {CodeObjectCacheService} {}
|
||||
|
||||
protected:
|
||||
friend class CodeObjectSerializeService;
|
||||
friend class NamedRegionObjectHandler;
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Async named region handling
|
||||
* @{ */
|
||||
/**
|
||||
* @brief The async named region jobs to handle.
|
||||
*
|
||||
* Only two, Code serialization goes in to a different queue.
|
||||
*/
|
||||
enum class NamedRegionJobType {
|
||||
JOB_ADD_NAMED_REGION,
|
||||
JOB_REMOVE_NAMED_REGION,
|
||||
};
|
||||
|
||||
class NamedRegionWorkItem {
|
||||
public:
|
||||
NamedRegionJobType GetType() const {
|
||||
return Type;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class WorkItemAddNamedRegion;
|
||||
NamedRegionWorkItem(NamedRegionJobType type)
|
||||
: Type {type} {}
|
||||
|
||||
private:
|
||||
NamedRegionJobType Type;
|
||||
};
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry} {}
|
||||
const fextl::string BaseFilename;
|
||||
const fextl::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
, Entry {std::move(entry)} {}
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
fextl::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
private:
|
||||
NamedRegionObjectHandler* NamedRegionHandler;
|
||||
CodeObjectSerializeService* CodeObjectCacheService;
|
||||
};
|
||||
|
||||
class NamedRegionObjectHandler final {
|
||||
public:
|
||||
NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
void HandleNamedRegionObjectJobs();
|
||||
|
||||
const CodeObjectSerializationConfig& GetDefaultSerializationConfig() const {
|
||||
return DefaultSerializationConfig;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
// Return a default code header based off the default serialization config
|
||||
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
|
||||
return CodeObjectSerializationHeader {
|
||||
.Config = DefaultSerializationConfig,
|
||||
.OriginalBase = Base,
|
||||
.OriginalOffset = Offset,
|
||||
.NumCodeEntries = 0,
|
||||
.NumRelocationsTo = 0,
|
||||
.TotalRelocationsCount = 0,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds an asynchronous add named region work item to the object queue
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion>(base, filename, executable, entry));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion>(Base, Size, std::move(Entry)));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
private:
|
||||
// Code version. If the code emission changes then this needs to increment
|
||||
constexpr static uint32_t CODE_VERSION = 0x0;
|
||||
|
||||
// Default cookie header for the file header
|
||||
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
|
||||
|
||||
// Code serialization config for our current process configuration
|
||||
CodeObjectSerializationConfig DefaultSerializationConfig;
|
||||
|
||||
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
|
||||
std::atomic<uint64_t> NamedWorkQueueJobs {};
|
||||
|
||||
// Mutex for ading new jobs to the work queue
|
||||
std::mutex NamedWorkQueueMutex {};
|
||||
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue {};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename, const fextl::string& filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Context specific code object serialization class
|
||||
*
|
||||
* Contains everything required for FEXCore to serialize code objects
|
||||
*/
|
||||
class CodeObjectSerializeService final {
|
||||
public:
|
||||
CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
/**
|
||||
* @brief Initialize the internal interface
|
||||
*
|
||||
* Is a public interface to allow the service to reinitialize after forking
|
||||
*/
|
||||
void Initialize();
|
||||
|
||||
/**
|
||||
* @brief Safely shut down the Code Object serialization service.
|
||||
*
|
||||
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
|
||||
*/
|
||||
void Shutdown();
|
||||
|
||||
/**
|
||||
* @name Async interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Loads a named region in to the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address that this named region is loaded
|
||||
* @param Size - The size of the region
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Unloads a named region from the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address of the named region
|
||||
* @param Size - The size of the region
|
||||
*/
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds a code object serialization job. As async as possible.
|
||||
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Synchronous interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Synchronously waits for this thread's job queue to become empty.
|
||||
*
|
||||
* This is necessary for when a thread is shutting down
|
||||
*
|
||||
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
|
||||
*/
|
||||
static void WaitForEmptyJobQueue(CodeSerializationMutex* ThreadJobRefCount) {
|
||||
// Once the shared mutex is empty this unique lock will be gained
|
||||
std::unique_lock lk {*ThreadJobRefCount};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Fetches object code from the Code Object Cache for JIT.
|
||||
*
|
||||
* @param GuestRIP - Which GuestRIP to search the cache for
|
||||
*
|
||||
* @return Data required for the JIT to relocate the Object code.
|
||||
*/
|
||||
const CodeObjectFileSection* FetchCodeObjectFromCache(uint64_t GuestRIP);
|
||||
/** @} */
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
/**
|
||||
* @brief Safely closes out code object regions from the map
|
||||
*
|
||||
* @param it - iterator to do a closure on
|
||||
*/
|
||||
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it);
|
||||
|
||||
CodeSerializationMutex& GetEntryMapMutex() {
|
||||
return EntryMapMutex;
|
||||
}
|
||||
CodeSerializationMutex& GetUnrelocatedEntryMapMutex() {
|
||||
return EntryMapMutex;
|
||||
}
|
||||
|
||||
CodeRegionMapType& GetEntryMap() {
|
||||
return AddressToEntryMap;
|
||||
}
|
||||
CodeRegionPtrMapType& GetUnrelocatedEntryMap() {
|
||||
return UnrelocatedAddressToEntryMap;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Notify the async thread that it has work to do
|
||||
*/
|
||||
void NotifyWork() {
|
||||
WorkAvailable.NotifyOne();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
Event WorkAvailable {};
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
// Mutex to hold when modifying the entry maps
|
||||
CodeSerializationMutex EntryMapMutex;
|
||||
CodeSerializationMutex UnrelocatedEntryMapMutex;
|
||||
|
||||
// Entry maps
|
||||
CodeRegionMapType AddressToEntryMap;
|
||||
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
|
||||
};
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -53,10 +53,11 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPRImmediate>},
|
||||
{0xC2, 2, &OpDispatchBuilder::RETOp},
|
||||
{0xC8, 1, &OpDispatchBuilder::EnterOp},
|
||||
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
|
||||
{0xCA, 2, &OpDispatchBuilder::RETFARIndirectOp},
|
||||
{0xCC, 2, &OpDispatchBuilder::INTOp},
|
||||
{0xCF, 1, &OpDispatchBuilder::IRETOp},
|
||||
{0xD7, 2, &OpDispatchBuilder::XLATOp},
|
||||
@@ -75,33 +76,4 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xFA, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xFC, 2, &OpDispatchBuilder::FLAGControlOp},
|
||||
};
|
||||
|
||||
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable_64[] = {
|
||||
{0x63, 1, &OpDispatchBuilder::MOVSXDOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
};
|
||||
|
||||
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable_32[] = {
|
||||
{0x06, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x07, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x0E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX>},
|
||||
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x17, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x1E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x1F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x27, 1, &OpDispatchBuilder::DAAOp},
|
||||
{0x2F, 1, &OpDispatchBuilder::DASOp},
|
||||
{0x37, 1, &OpDispatchBuilder::AAAOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::AASOp},
|
||||
{0x40, 8, &OpDispatchBuilder::INCOp},
|
||||
{0x48, 8, &OpDispatchBuilder::DECOp},
|
||||
|
||||
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
|
||||
{0x61, 1, &OpDispatchBuilder::POPAOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
{0xCE, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xD4, 1, &OpDispatchBuilder::AAMOp},
|
||||
{0xD5, 1, &OpDispatchBuilder::AADOp},
|
||||
{0xD6, 1, &OpDispatchBuilder::SALCOp},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -19,8 +19,12 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
@@ -32,24 +36,32 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
@@ -58,13 +70,17 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
// The result is swizzled differently than expected
|
||||
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref Result {};
|
||||
Ref ConstantVector {};
|
||||
@@ -96,21 +112,29 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Result = _VSha256U0(Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
@@ -118,22 +142,16 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VSha256U1(Src1, Src2);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
// Returns whether at least 2/3 of A/B/C is true.
|
||||
// Expressed as (A & (B | C)) | (B & C)
|
||||
//
|
||||
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
|
||||
auto And = _And(OpSize::i32Bit, B, C);
|
||||
auto Or = _Or(OpSize::i32Bit, B, C);
|
||||
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
@@ -159,101 +177,121 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
auto Result = shuffle_abcd(A, B);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
@@ -261,28 +299,41 @@ Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -201,7 +201,7 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Again 64-bit as masking is more expensive given our ConstProp design.
|
||||
// Again 64-bit as masking is more expensive.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
@@ -263,12 +263,10 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
|
||||
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
// If CF not inverted, we use .cc since the increment happens when the
|
||||
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
return _NZCVSelectIncrement(OpSize, CFInverted ? CondClass::UGE : CondClass::ULT, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref Res;
|
||||
|
||||
@@ -288,11 +286,11 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Ref Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Res, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -304,8 +302,6 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
@@ -325,10 +321,10 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Src1, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -349,10 +345,10 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _SubWithFlags(SrcSize, Src1, Src2);
|
||||
Res = SubWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_SubNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Sub(OpSize::i32Bit, Src1, Src2);
|
||||
Res = Sub(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -379,10 +375,10 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _AddWithFlags(SrcSize, Src1, Src2);
|
||||
Res = AddWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_AddNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Add(OpSize::i32Bit, Src1, Src2);
|
||||
Res = Add(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -410,7 +406,7 @@ void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
// undefined, this does what we need after inverting carry.
|
||||
auto Zero = _InlineConstant(0);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
@@ -427,7 +423,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
|
||||
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ namespace FEXCore::IR {
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
@@ -71,9 +72,28 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xC8), 1, &OpDispatchBuilder::SHA1NEXTEOp},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, &OpDispatchBuilder::SHA1MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, &OpDispatchBuilder::SHA1MSG2Op},
|
||||
{OPD(PF_38_NONE, 0xCB), 1, &OpDispatchBuilder::SHA256RNDS2Op},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
|
||||
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
|
||||
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
|
||||
{OPD(PF_38_66, 0xDF), 1, &OpDispatchBuilder::AESDecLastOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
{OPD(PF_38_66, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
|
||||
{OPD(PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
{OPD(PF_38_F3, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
};
|
||||
|
||||
@@ -29,6 +29,7 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
@@ -36,6 +37,8 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
@@ -65,11 +68,6 @@ constexpr DispatchTableEntry OpDispatch_H0F3ATableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
|
||||
|
||||
@@ -117,7 +117,9 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, &OpDispatchBuilder::INCOp}, // INC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, &OpDispatchBuilder::CALLAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 3), 1, &OpDispatchBuilder::CALLFARIndirectOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, &OpDispatchBuilder::JUMPAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 5), 1, &OpDispatchBuilder::JUMPFARIndirectOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp},
|
||||
|
||||
// GROUP 11
|
||||
|
||||
@@ -69,10 +69,16 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 7), 1, &OpDispatchBuilder::RDPIDOp},
|
||||
|
||||
// GROUP 12
|
||||
@@ -145,6 +151,9 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
// GROUP 17
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_17, PF_66, 0), 1, &OpDispatchBuilder::Extrq_imm},
|
||||
|
||||
// GROUP P
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
@@ -156,18 +165,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables_64[] = {
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -17,6 +17,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryModRMTables[] = {
|
||||
// REG /7
|
||||
{((3 << 3) | 0), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
{((3 << 3) | 4), 1, &OpDispatchBuilder::CLZeroOp},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -145,7 +145,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3E, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
#endif
|
||||
};
|
||||
@@ -198,6 +198,8 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
|
||||
{0x79, 1, &OpDispatchBuilder::Insertq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
@@ -256,6 +258,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x79, 1, &OpDispatchBuilder::Extrq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
@@ -313,20 +316,4 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_TwoByteOpTable_64[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SyscallOp, true>},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_TwoByteOpTable_32[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::NOPOp},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
File diff suppressed because it is too large.
Load diff
Loaded 100 of 986 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user