mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 00:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
da069571f3 | ||
|
|
d2bac45b49 | ||
|
|
8913c59acc | ||
|
|
c3261b4aeb | ||
|
|
c7fb95aec5 | ||
|
|
c00cef6dc1 | ||
|
|
429ff94dc5 | ||
|
|
a6c67ca749 | ||
|
|
f51812a670 | ||
|
|
686294f1c4 | ||
|
|
a47ed105e7 | ||
|
|
b2d579a268 | ||
|
|
eb1050092f | ||
|
|
b3794f5541 | ||
|
|
1ecfa3253d | ||
|
|
5daf007b6a | ||
|
|
8efa5febd0 | ||
|
|
5fee8028cd | ||
|
|
6bc7a83c64 | ||
|
|
e55b5d0d11 | ||
|
|
19de7f2785 | ||
|
|
e32c5384ab | ||
|
|
b391fe6b92 | ||
|
|
4cfb81156f | ||
|
|
6121708e55 | ||
|
|
6ab214adea | ||
|
|
90b1ac4162 | ||
|
|
3abe6c14a1 | ||
|
|
fc1b500eff | ||
|
|
5d47b9195b | ||
|
|
a8272b74f6 | ||
|
|
12dc16780f | ||
|
|
b8af569841 | ||
|
|
8bee101795 | ||
|
|
2d66bc258a | ||
|
|
d2f86e49f7 | ||
|
|
d66cd16cfb | ||
|
|
04e785e434 | ||
|
|
15a1a0f7d9 | ||
|
|
0a58ce6134 | ||
|
|
8f5607f0e8 | ||
|
|
a21789d3d8 | ||
|
|
efd6e95059 | ||
|
|
9bdb1f4306 | ||
|
|
ae4b7135d5 | ||
|
|
d503366816 | ||
|
|
0fe2827fcc | ||
|
|
cd6722f77b | ||
|
|
ffb745b662 | ||
|
|
bb10f25808 | ||
|
|
aa1076d12b | ||
|
|
1e827ec7a6 | ||
|
|
3fe2650787 | ||
|
|
3e99e814bc | ||
|
|
2019f8138e | ||
|
|
e44d1f136b | ||
|
|
09872402df | ||
|
|
b078a41a02 | ||
|
|
3a5eeb5700 | ||
|
|
9433ae3405 | ||
|
|
4658b24f9a | ||
|
|
4e7d0e6be0 | ||
|
|
4ddd98708f | ||
|
|
c161fd218c | ||
|
|
7e257cc268 | ||
|
|
d8ef70280c | ||
|
|
ec003281be | ||
|
|
e58f67b76c | ||
|
|
57178abcd2 | ||
|
|
73ca4f8314 | ||
|
|
527752c25b | ||
|
|
38fa866c91 | ||
|
|
c902b8807a | ||
|
|
4934c1fd94 | ||
|
|
766fbe3db3 | ||
|
|
29405f2690 | ||
|
|
51f505acca | ||
|
|
77415538f7 | ||
|
|
9fb69ed206 | ||
|
|
735a4f90db | ||
|
|
7ef8dc13ba | ||
|
|
656477ec63 | ||
|
|
d080180e85 | ||
|
|
af1d2d6005 | ||
|
|
90c1282f3a | ||
|
|
d5d7eec8b0 | ||
|
|
5337b9537d | ||
|
|
e72c016230 | ||
|
|
072cf4c5bd | ||
|
|
27ededf47f | ||
|
|
82d7f9fdd7 | ||
|
|
d85153d6b3 | ||
|
|
6b698e6cd1 | ||
|
|
9475f79ec6 | ||
|
|
f906c6a0f4 | ||
|
|
e88c92de57 | ||
|
|
48ed906a7b | ||
|
|
b03b02d2f2 | ||
|
|
d00d476a0a | ||
|
|
ac1e32994a | ||
|
|
800d447f3d | ||
|
|
8111b7cc7f | ||
|
|
a86c922073 | ||
|
|
46fb8583bb | ||
|
|
3487d120ec | ||
|
|
07394d6a6e | ||
|
|
8d3204171c | ||
|
|
7641f722e9 | ||
|
|
b51fa497c5 | ||
|
|
34722bed3d | ||
|
|
981c3009ee | ||
|
|
38cf357d85 | ||
|
|
2533ed4a63 | ||
|
|
5a4691fdfc | ||
|
|
6c035a0d61 | ||
|
|
a234aa300d | ||
|
|
f6abbedbd1 | ||
|
|
b6fe4cd6dd | ||
|
|
bdae4f6915 | ||
|
|
f8b6edfb2b | ||
|
|
0a1ecdf6ae | ||
|
|
beec203f56 | ||
|
|
d323032ec9 | ||
|
|
7472b21f33 | ||
|
|
1058575d3a | ||
|
|
e9867ca35a | ||
|
|
84277319fa | ||
|
|
8f8aa55c7f | ||
|
|
72a4063651 | ||
|
|
0b1229da55 | ||
|
|
1d3ce30e50 | ||
|
|
71187d3ad7 | ||
|
|
dd8a3a9aea | ||
|
|
7b2fc37651 | ||
|
|
426569d74d | ||
|
|
e877d5b82c | ||
|
|
572e0d04d5 | ||
|
|
e3d7161ac5 | ||
|
|
20caf69951 | ||
|
|
fcbf0de05a | ||
|
|
731e4d6271 | ||
|
|
48beb18f29 | ||
|
|
41c8731443 | ||
|
|
efb276f489 | ||
|
|
9febddefa3 | ||
|
|
ce9a860335 | ||
|
|
0123946ed1 | ||
|
|
c7098d0da1 | ||
|
|
65a162bdf9 | ||
|
|
2de485d02a | ||
|
|
bf64facaf6 | ||
|
|
2e7fc60dbf | ||
|
|
baddfe00b1 | ||
|
|
e7e59204d3 | ||
|
|
95d5b14f99 | ||
|
|
802eaee9c8 | ||
|
|
e89f48f237 | ||
|
|
e771e25632 | ||
|
|
25c202575e | ||
|
|
f59fc0f747 | ||
|
|
f7a076e00c | ||
|
|
56fadecdaf | ||
|
|
9f681f9e41 | ||
|
|
1b11f2f184 | ||
|
|
b2e61c37be | ||
|
|
7c0cf51f09 | ||
|
|
649a49488b | ||
|
|
aa2180d494 | ||
|
|
969cae581c | ||
|
|
1bf7e2544a | ||
|
|
fad22144a2 | ||
|
|
b440e176fb | ||
|
|
56c6b0d2cb | ||
|
|
0596a963e1 | ||
|
|
357cc04940 | ||
|
|
7c6e836865 | ||
|
|
54a7317312 | ||
|
|
1fb20710e6 | ||
|
|
811ea093b5 | ||
|
|
71fe9aee21 | ||
|
|
38c834e731 | ||
|
|
740ff60a71 | ||
|
|
ee69b9f650 | ||
|
|
a4565ce783 | ||
|
|
3131ee4de1 | ||
|
|
3ecc66fbcf | ||
|
|
e322e84785 | ||
|
|
f0fa7a5b6a | ||
|
|
718221be71 | ||
|
|
ee592ba03c | ||
|
|
56947f3a94 | ||
|
|
f41b9bc514 | ||
|
|
0463512c6c | ||
|
|
47369d058e | ||
|
|
b6f34fa209 | ||
|
|
03e0ca9833 | ||
|
|
d19473160d | ||
|
|
aeb2c98cbf | ||
|
|
2350ae5a07 | ||
|
|
6076d1747e | ||
|
|
f4ce6fb621 | ||
|
|
f0d25d413b | ||
|
|
c84503c271 | ||
|
|
1465df874b | ||
|
|
60c52e3826 | ||
|
|
13b806130b | ||
|
|
22058c06a1 | ||
|
|
d5c96555f1 | ||
|
|
c3e8cd8d30 | ||
|
|
d761fc44f4 | ||
|
|
a1aa2547ce | ||
|
|
44213c3968 | ||
|
|
830bd347c5 | ||
|
|
bcfdf39d63 | ||
|
|
bfed21870f | ||
|
|
4278c48791 | ||
|
|
c1db7a78b1 | ||
|
|
1bf06f8946 | ||
|
|
09bfe58827 | ||
|
|
474c780399 | ||
|
|
4a67893f1d | ||
|
|
73ffaa1e18 | ||
|
|
e675f4241a | ||
|
|
7a61d9d2b4 | ||
|
|
06b950a9cd | ||
|
|
de70651406 | ||
|
|
5ad7fdb2f3 | ||
|
|
9b6cc8f7e0 | ||
|
|
5c6de4ed14 | ||
|
|
82f936cb6d | ||
|
|
493b952e3f | ||
|
|
460a21625e | ||
|
|
00ab3f8440 | ||
|
|
034b62292b | ||
|
|
7b615a07d0 | ||
|
|
704841f004 | ||
|
|
c122f3faf9 | ||
|
|
f74f276d64 | ||
|
|
4b10cbdafd | ||
|
|
04c701e912 | ||
|
|
0c29f8faad | ||
|
|
5ed82fa0f6 | ||
|
|
6cca007817 | ||
|
|
c0a9463700 | ||
|
|
55b3d67eb4 | ||
|
|
65ddae1b71 | ||
|
|
063f524084 | ||
|
|
2dd0a82059 | ||
|
|
b810070e9f | ||
|
|
4c7ac17f7d | ||
|
|
84767c8b20 | ||
|
|
eccfb53bd5 | ||
|
|
f4e930262f | ||
|
|
d26d9e7e03 | ||
|
|
51c1998d70 | ||
|
|
87f818249d | ||
|
|
d81f92f5e2 | ||
|
|
29ffe02afe | ||
|
|
dfe4076fe4 | ||
|
|
44f9df062e | ||
|
|
6f98ef8cbb | ||
|
|
d54888a4c6 | ||
|
|
38c58706da | ||
|
|
8ad9286bd4 | ||
|
|
03bc962564 | ||
|
|
947b7ae6fe | ||
|
|
30317ac979 | ||
|
|
4544e7c1af | ||
|
|
34050431ff | ||
|
|
65439956bf | ||
|
|
a6cbce4fd7 | ||
|
|
17b851d4f3 | ||
|
|
362b5728be | ||
|
|
bd5159c7d5 | ||
|
|
660dfcd1f9 | ||
|
|
4e21177988 | ||
|
|
5a83a65905 | ||
|
|
061fc44923 | ||
|
|
0c7afa0672 | ||
|
|
30bf0d5767 | ||
|
|
ee8e3127d2 | ||
|
|
e8f64f2976 | ||
|
|
0a4b21da87 | ||
|
|
074777bc75 | ||
|
|
47c403998a | ||
|
|
0fa095cec5 | ||
|
|
cfa4e7f165 | ||
|
|
75b8226e7d | ||
|
|
9e3c50ca2c | ||
|
|
365d8b9508 | ||
|
|
02ebe06496 | ||
|
|
34d5e70e6b | ||
|
|
2d8bd7b59d | ||
|
|
3a6f5e638b | ||
|
|
c06274066f | ||
|
|
820b0be9f2 | ||
|
|
3e74817dd9 | ||
|
|
3b9a2d2141 | ||
|
|
9ba431a51d | ||
|
|
d0d2229db6 | ||
|
|
286258a2f2 | ||
|
|
ac14a88647 | ||
|
|
bf19578673 | ||
|
|
c5f396d889 | ||
|
|
22325500d9 | ||
|
|
51eede080c | ||
|
|
503c86d47d | ||
|
|
0e0887181d | ||
|
|
974cc591bd | ||
|
|
844afdb653 | ||
|
|
c3c643d9b7 | ||
|
|
b51b0c5f62 | ||
|
|
96bdefddc9 | ||
|
|
ba74a6a252 | ||
|
|
79427097e1 | ||
|
|
7ec21b7121 | ||
|
|
29115b3185 | ||
|
|
33b3814642 | ||
|
|
85431a8132 | ||
|
|
2d9ef56a8b | ||
|
|
b2ae829731 | ||
|
|
a3c544a9a1 | ||
|
|
9c2292289b | ||
|
|
b514548ca2 | ||
|
|
692a4a8fcd | ||
|
|
3a07cf7d70 | ||
|
|
7f5421fc26 | ||
|
|
04d62cd269 | ||
|
|
ac55e468a7 | ||
|
|
e20db7dc88 | ||
|
|
235bee9191 | ||
|
|
7a429b01c7 | ||
|
|
fbf5b14933 | ||
|
|
0dfd5dd96f | ||
|
|
6d5acec958 | ||
|
|
beed43e577 | ||
|
|
fc04b9113e | ||
|
|
68c038085a | ||
|
|
176f5a2860 | ||
|
|
e7c9623aa9 | ||
|
|
53ca2ac378 | ||
|
|
d140cb4450 | ||
|
|
0ba501636e | ||
|
|
ceca9fff17 | ||
|
|
3e31abb645 | ||
|
|
6c07cd319b | ||
|
|
5247b7124f | ||
|
|
6cec557855 | ||
|
|
1868bd6777 | ||
|
|
7c7efeda82 | ||
|
|
414486f1dd | ||
|
|
b8bc9659d4 | ||
|
|
d52a6e6fc4 | ||
|
|
5ab41056ab | ||
|
|
0692b34192 | ||
|
|
3636c332ff | ||
|
|
cb18963ded | ||
|
|
8cf92d3303 | ||
|
|
cdc5c15b4b | ||
|
|
ed313edd07 | ||
|
|
9b981a4f61 | ||
|
|
c791893b4a | ||
|
|
2605c7e0b3 | ||
|
|
f315948028 | ||
|
|
869367f7e2 | ||
|
|
3a2c7e8edd | ||
|
|
0a34a43976 | ||
|
|
78cd21d78f | ||
|
|
9f18de0196 | ||
|
|
0bffdc4e27 | ||
|
|
0f5ff53386 | ||
|
|
21611fc1ad | ||
|
|
99e1eb5452 | ||
|
|
8c3ca44c57 | ||
|
|
7a85e17d14 | ||
|
|
b533dcd86d | ||
|
|
a379ce6fed | ||
|
|
efd5c51110 | ||
|
|
886db4ffca | ||
|
|
e6f6ee2bcd | ||
|
|
4a6b5d4ec7 | ||
|
|
027e7624cb | ||
|
|
0e31077735 | ||
|
|
c8a9dd0d0a | ||
|
|
2bd7ddaa31 | ||
|
|
5566b4455b | ||
|
|
fed2c13521 | ||
|
|
37d092aab8 | ||
|
|
5626f4e50a | ||
|
|
efbc42dac3 | ||
|
|
160934884d | ||
|
|
af1cfcb9bd | ||
|
|
d6f726fc23 | ||
|
|
92ee071eb2 | ||
|
|
bdfa8ad4f3 | ||
|
|
37540f4927 | ||
|
|
000ab5ff19 | ||
|
|
f054274948 | ||
|
|
081907e168 | ||
|
|
1a115a8ce6 | ||
|
|
fd9158c75f | ||
|
|
1bde30a196 | ||
|
|
764aacaa8f | ||
|
|
a848211926 | ||
|
|
f1a42869d5 | ||
|
|
97a6ba9931 | ||
|
|
f4744f1e79 | ||
|
|
321f686108 | ||
|
|
90340350fa | ||
|
|
6f4fd4467b | ||
|
|
b31ce13f68 | ||
|
|
260d3b0b4e | ||
|
|
c8c7ffbf05 | ||
|
|
4b03185b77 | ||
|
|
52ec572db3 | ||
|
|
dc31cf83c6 | ||
|
|
8a4f51257d | ||
|
|
051469fa16 | ||
|
|
3f6cdc2e03 | ||
|
|
f3449f2b00 | ||
|
|
d7691d9a25 | ||
|
|
f414d4934c | ||
|
|
014917301a | ||
|
|
cc483acbde | ||
|
|
07f8a4eadd | ||
|
|
5fd127b53a | ||
|
|
ece89ddeab | ||
|
|
a1565a7d99 | ||
|
|
2f9b0de742 | ||
|
|
7e5f1b5859 | ||
|
|
40fd4bbb66 | ||
|
|
e4143352c9 | ||
|
|
c045e14837 | ||
|
|
f0f3c215ce | ||
|
|
8f4113d859 | ||
|
|
4cfc2ac1a4 | ||
|
|
d2aa5217dc | ||
|
|
e8baf4a28c | ||
|
|
e438d32879 | ||
|
|
32ef10b273 | ||
|
|
ad296051b7 | ||
|
|
e603136918 | ||
|
|
f8a61f7d7e | ||
|
|
cb5ba8baae | ||
|
|
079e70fc4e | ||
|
|
3d701f5fcf | ||
|
|
b31e4a3c27 | ||
|
|
992d6e8477 | ||
|
|
f6cdb165a3 | ||
|
|
f60388d160 | ||
|
|
96fa2ad8eb | ||
|
|
f143462ebe | ||
|
|
51fa61a1cd | ||
|
|
048e967546 | ||
|
|
608fd49ac3 | ||
|
|
bb630797b5 | ||
|
|
01a6e914f2 | ||
|
|
0190e1a00b | ||
|
|
11a87c22f9 | ||
|
|
5f6c0d2245 | ||
|
|
caaacb6c15 | ||
|
|
767c61c08b | ||
|
|
dc93e30451 | ||
|
|
368162df87 | ||
|
|
9eb2106ed2 | ||
|
|
cfc05b78fe | ||
|
|
d2a42c0038 | ||
|
|
58a3d174ec | ||
|
|
9c605e7333 | ||
|
|
1fd7e88ffd | ||
|
|
c5e7da0631 | ||
|
|
68f58e415f | ||
|
|
698abec25c | ||
|
|
1578f5ed47 | ||
|
|
80d7b5a5c9 | ||
|
|
eb023ceb51 | ||
|
|
5d1fda7d7f | ||
|
|
fafc04a59e | ||
|
|
ddcca58f64 | ||
|
|
fe8f5c745d | ||
|
|
d876224358 | ||
|
|
c4306f2f0a | ||
|
|
8b7a227820 | ||
|
|
d7afcee622 | ||
|
|
a421ff1105 | ||
|
|
5997030c97 | ||
|
|
3c8086373b | ||
|
|
10ec6b63b6 | ||
|
|
49087007be | ||
|
|
0897cd8777 | ||
|
|
4b945a9041 | ||
|
|
d66ed71bc6 | ||
|
|
b5b34df155 | ||
|
|
ff51435747 | ||
|
|
def561986b | ||
|
|
5c258d4a2a | ||
|
|
0d53f2b45c | ||
|
|
4f03044fe7 | ||
|
|
b967538435 | ||
|
|
e53f3969e9 | ||
|
|
5026bf8247 | ||
|
|
09cb4f5fc5 | ||
|
|
ddd7a550e4 | ||
|
|
3398f22c16 | ||
|
|
7f17519fbf | ||
|
|
a65884f9ae | ||
|
|
d70766f4c8 | ||
|
|
1365aa8881 | ||
|
|
fe5bc02682 | ||
|
|
6f096e7c4b | ||
|
|
389ad737e6 | ||
|
|
c00f7813a2 | ||
|
|
e5ceaa182d | ||
|
|
eeb8eb1824 | ||
|
|
7c6444c37c | ||
|
|
4a179c8f87 | ||
|
|
2b3895a514 | ||
|
|
0bb0f9cec7 | ||
|
|
6a07ea73a8 | ||
|
|
5c51c54ccc | ||
|
|
cac3767c20 | ||
|
|
e82d2c70c3 | ||
|
|
9716fc73d3 | ||
|
|
aaa8eef9f1 | ||
|
|
c036868938 | ||
|
|
2726f35a10 | ||
|
|
c740801ea5 | ||
|
|
407dc5d4dd | ||
|
|
7dda75646d | ||
|
|
b89f5b8a03 | ||
|
|
16869d8954 | ||
|
|
e353ae8408 | ||
|
|
aa3c963df1 | ||
|
|
0c8cfa7bb2 | ||
|
|
dd837fa693 | ||
|
|
eeff198ff1 | ||
|
|
0cd6371c1b | ||
|
|
5aff16f72b | ||
|
|
2c7896bce4 | ||
|
|
6dd72a09e5 | ||
|
|
4e6a4d9b69 | ||
|
|
44484a7b05 | ||
|
|
6ac7ba388f | ||
|
|
36d0a67070 | ||
|
|
0650dd1992 | ||
|
|
e2d58809ed | ||
|
|
967a74cda9 | ||
|
|
6cc6181261 | ||
|
|
f175b525f4 | ||
|
|
319f1e66cb | ||
|
|
d16969bdce | ||
|
|
c17858bfeb | ||
|
|
ec24fc3d5d | ||
|
|
8819fa88d5 | ||
|
|
5e9c2110db | ||
|
|
5aaa18e7a2 | ||
|
|
8a551b9e64 | ||
|
|
aa548bd19c | ||
|
|
9565f16d84 | ||
|
|
eb76dbdf4d | ||
|
|
967c04e252 | ||
|
|
49de1fac59 | ||
|
|
3c205eb35e | ||
|
|
682b8ef705 | ||
|
|
8c47a40625 | ||
|
|
ee5c6af868 | ||
|
|
1670c89b5e | ||
|
|
e9d34c7ded | ||
|
|
c2f2c58367 | ||
|
|
38acd9e18c | ||
|
|
21604708ca | ||
|
|
5fec6faeb6 | ||
|
|
b659701cef | ||
|
|
e902ad5278 | ||
|
|
0f54369f9b | ||
|
|
f640dcc7a4 | ||
|
|
b59da0e049 | ||
|
|
4ae3aef502 | ||
|
|
926fa3c24c | ||
|
|
8c1740f592 | ||
|
|
611aa0a5b9 | ||
|
|
0b8b5108b9 | ||
|
|
e409a0afec | ||
|
|
24211f8523 | ||
|
|
94462e4dd0 | ||
|
|
aa44668a41 | ||
|
|
33a675624c | ||
|
|
5745b419d9 | ||
|
|
9100235041 | ||
|
|
4d62e75d5a | ||
|
|
b6d3df0e54 | ||
|
|
d5d1230422 | ||
|
|
f35a212f06 | ||
|
|
f49d82deb3 | ||
|
|
77b801b8ae | ||
|
|
163a3fc899 | ||
|
|
9b627e8743 | ||
|
|
9d6865f62d | ||
|
|
d547f2b8f2 | ||
|
|
b5c9eb3463 | ||
|
|
00a3793814 | ||
|
|
4f8d7cb89c | ||
|
|
9310ac59f4 | ||
|
|
9e4de3bfe8 | ||
|
|
598b99fe58 | ||
|
|
9dda76ebe6 | ||
|
|
f5def7ae1c | ||
|
|
58bcf691d6 | ||
|
|
33186b803d | ||
|
|
bb1d7d0750 | ||
|
|
4897c1e80c | ||
|
|
e7605c94dd | ||
|
|
ab5a10374b | ||
|
|
dc2a26582f | ||
|
|
e6512fbe4d | ||
|
|
dbca441573 | ||
|
|
31adc953b7 | ||
|
|
0dd7f5e2f7 | ||
|
|
36556c4705 | ||
|
|
e29fac3f25 | ||
|
|
c984bdb42a | ||
|
|
2645b374a7 | ||
|
|
10192eebe5 | ||
|
|
0e57cbf5b9 | ||
|
|
c7413d96ad | ||
|
|
f3ba8cb33c | ||
|
|
e714933b10 | ||
|
|
9a0f83cbf6 | ||
|
|
c4d30e8437 | ||
|
|
7b30df8a56 | ||
|
|
86aa459dd1 | ||
|
|
6c9a47fc71 | ||
|
|
7a72cf631b | ||
|
|
3d98eef7b8 | ||
|
|
f2011b0b79 | ||
|
|
c0b8a0d9c3 | ||
|
|
5bd0afa307 | ||
|
|
20fb0da7d9 | ||
|
|
35cb1f710c | ||
|
|
7c7b99f9c5 | ||
|
|
09879bd962 | ||
|
|
c8f3fe3788 | ||
|
|
88e3164b55 | ||
|
|
bebd7400c0 | ||
|
|
e190d029dc | ||
|
|
37a70e2ec6 | ||
|
|
6ff073ac11 | ||
|
|
b6e4c47abc | ||
|
|
daf8b409f3 | ||
|
|
0a35029d9a | ||
|
|
e8b51ac0a4 | ||
|
|
f20e626fda | ||
|
|
2672e06aa5 | ||
|
|
4a2b4be59d | ||
|
|
05be9445cd | ||
|
|
4cb4c0e277 | ||
|
|
0135e2d78c | ||
|
|
1266a5a5ce | ||
|
|
7255850e9d | ||
|
|
002a8bebd8 | ||
|
|
452c9fd19b | ||
|
|
0b0e15c9b4 | ||
|
|
ae9db336e7 | ||
|
|
360ea538b7 | ||
|
|
f99900bdc5 | ||
|
|
34f2bce203 | ||
|
|
a05d6ae082 | ||
|
|
6a4eb434f7 | ||
|
|
0c3eb41f57 | ||
|
|
e4bfbda008 | ||
|
|
b4c797c199 | ||
|
|
ebae1bf71d | ||
|
|
fa293e5e8f | ||
|
|
720c89e648 | ||
|
|
abc5f5abd4 | ||
|
|
22ab16b217 | ||
|
|
b80f05dd86 | ||
|
|
1642a0f4d9 | ||
|
|
96e990dade | ||
|
|
3423a12ab6 | ||
|
|
c4a0fb6609 | ||
|
|
8cf23eba9c | ||
|
|
fbe817f1c4 | ||
|
|
3b61394548 | ||
|
|
8ed6b36181 | ||
|
|
9141322666 | ||
|
|
2d91c5441e | ||
|
|
2ff1589c35 | ||
|
|
a308b9edbe | ||
|
|
f6a8e595df | ||
|
|
eaaf62e4b7 | ||
|
|
dc5fc57e19 | ||
|
|
c3c0bbb060 | ||
|
|
98a91d242d | ||
|
|
8da6b1e729 | ||
|
|
d260bce31e | ||
|
|
a94aaca0ec | ||
|
|
2ac3ca3ead | ||
|
|
c559445b62 | ||
|
|
8ea4e4f663 | ||
|
|
2da700fe82 | ||
|
|
0a32ba9b29 | ||
|
|
f51832ca6b | ||
|
|
5f9a30af7a | ||
|
|
453ef0b94c | ||
|
|
4779e64cf5 | ||
|
|
e6025cc087 | ||
|
|
cd4c224dc0 | ||
|
|
07f117b957 | ||
|
|
36e4f9a070 | ||
|
|
67c751d3e5 | ||
|
|
1c59bfeb9f | ||
|
|
2b0041a291 | ||
|
|
39cc1e5116 | ||
|
|
f46856c090 | ||
|
|
b7859052da | ||
|
|
3b30d8103f | ||
|
|
d19fc36f8f | ||
|
|
5fbf76ee9e | ||
|
|
5f79761fe4 | ||
|
|
621af9e7e8 | ||
|
|
f4d107c8f0 | ||
|
|
59643db331 | ||
|
|
46f36dc89d | ||
|
|
384882744c | ||
|
|
b36d1f7e7b | ||
|
|
dd7c66db2e | ||
|
|
47ac9edd7a | ||
|
|
6e6d640fec | ||
|
|
eb88366614 | ||
|
|
50d6ddd591 | ||
|
|
249de7c758 | ||
|
|
c350888d70 | ||
|
|
f23bef1653 | ||
|
|
d17f33a922 | ||
|
|
50a3ca0d6d | ||
|
|
8745455a5b | ||
|
|
e9ab514962 | ||
|
|
4eb0948451 | ||
|
|
8d9f19bd73 | ||
|
|
9fe7bc5818 | ||
|
|
5a590a9b11 | ||
|
|
a4acd64246 | ||
|
|
304b5de1af | ||
|
|
ca2fe0b301 | ||
|
|
bbef4d762c | ||
|
|
f77841d784 | ||
|
|
219a4777c6 | ||
|
|
29f0e9b4d6 | ||
|
|
b75efc42bf | ||
|
|
90dbd4766d | ||
|
|
f7b911ca43 | ||
|
|
6de0333708 | ||
|
|
ab5d3ab22b | ||
|
|
355c3428c0 | ||
|
|
c76b7bfe8d | ||
|
|
64c5362580 | ||
|
|
77469de247 | ||
|
|
48fd827004 | ||
|
|
f09d511ac8 |
No files matched your search
+1
-1
@@ -4,7 +4,7 @@ compile_commands.json
|
||||
vim_rc
|
||||
Config.json
|
||||
|
||||
[Bb]uild*/
|
||||
[Bb]uild*
|
||||
[Bb]in/
|
||||
out/
|
||||
.vscode/
|
||||
|
||||
@@ -5,9 +5,6 @@
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/imgui"]
|
||||
path = External/imgui
|
||||
url = https://github.com/Sonicadvance1/imgui.git
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
|
||||
+35
-22
@@ -7,7 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
set(USE_FEXCONFIG_TOOLKIT "imgui" CACHE STRING "If set, build FEXConfig (qt or imgui)")
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -37,6 +37,7 @@ option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
@@ -232,7 +233,6 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
# The glibc jemalloc subproject which hooks the glibc allocator.
|
||||
# Required for thunks to work.
|
||||
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
|
||||
add_definitions(-DENABLE_JEMALLOC_GLIBC=1)
|
||||
add_subdirectory(External/jemalloc_glibc/)
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
@@ -244,9 +244,7 @@ endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
# The jemalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
" jemalloc disabled!\n"
|
||||
@@ -273,8 +271,10 @@ if (BUILD_TESTS)
|
||||
set(COMPILE_VIXL_DISASSEMBLER TRUE)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
if (COMPILE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
@@ -284,29 +284,37 @@ endif()
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
|
||||
pkg_search_module(xxhash IMPORTED_TARGET xxhash libxxhash)
|
||||
if (TARGET PkgConfig::xxhash AND NOT CMAKE_CROSSCOMPILING)
|
||||
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
|
||||
else()
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
endif()
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(External/Catch2/)
|
||||
find_package(Catch2 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
endif()
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
if (USE_FEXCONFIG_TOOLKIT STREQUAL "imgui")
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
@@ -405,10 +413,13 @@ configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
|
||||
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTS)
|
||||
include(CTest)
|
||||
enable_testing()
|
||||
message(STATUS "Unit tests are enabled")
|
||||
if (NOT BUILD_TESTING)
|
||||
# CMake checks this variable before generating CTestTestfile.cmake
|
||||
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
|
||||
endif()
|
||||
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
@@ -423,7 +434,7 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
@@ -469,6 +480,7 @@ if (BUILD_THUNKS)
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
@@ -487,6 +499,7 @@ if (BUILD_THUNKS)
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
|
||||
@@ -301,8 +301,9 @@ public:
|
||||
xbfiz_helper(true, s, rd, rn, lsb, width);
|
||||
}
|
||||
void asr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
sbfm(s, rd, rn, shift, RegSize_m1);
|
||||
}
|
||||
|
||||
void uxtb(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
@@ -325,14 +326,14 @@ public:
|
||||
}
|
||||
|
||||
void lsl(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
ubfm(s, rd, rn, (RegSize - shift) % RegSize, RegSize - shift - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
ubfm(s, rd, rn, (RegSizeInBits(s) - shift) & RegSize_m1, RegSize_m1 - shift);
|
||||
}
|
||||
void lsr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsr a region larger than the register");
|
||||
ubfm(s, rd, rn, shift, RegSize - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
ubfm(s, rd, rn, shift, RegSize_m1);
|
||||
}
|
||||
void ubfx(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
LOGMAN_THROW_A_FMT(width > 0, "ubfx needs width > 0");
|
||||
@@ -368,6 +369,7 @@ public:
|
||||
}
|
||||
|
||||
void ror(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t Imm) {
|
||||
Imm &= RegSizeInBits(s) - 1;
|
||||
extr(s, rd, rn, rn, Imm);
|
||||
}
|
||||
|
||||
|
||||
@@ -1070,7 +1070,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10110, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1082,7 +1082,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10110, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1095,7 +1095,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10111, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1107,7 +1107,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10111, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1123,7 +1123,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11000, rd, rn);
|
||||
}
|
||||
@@ -1138,7 +1138,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11001, rd, rn);
|
||||
}
|
||||
@@ -1154,7 +1154,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
@@ -1169,7 +1169,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
@@ -1184,7 +1184,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
@@ -1199,7 +1199,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -1214,7 +1214,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11110, rd, rn);
|
||||
}
|
||||
@@ -1229,7 +1229,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11111, rd, rn);
|
||||
}
|
||||
|
||||
@@ -354,6 +354,7 @@ enum class SystemRegister : uint32_t {
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
|
||||
TPIDRRO_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b011>(),
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
|
||||
};
|
||||
|
||||
@@ -6,4 +6,3 @@ mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xf
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
expose_interpreter optional
|
||||
@@ -6,4 +6,3 @@ mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xf
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
expose_interpreter optional
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 31aa7f634b...29f979ee5a.
Vendored
+1
-1
Submodule External/drm-headers updated: 34a20394f7...8efb6dc03f.
Vendored
+1
-1
Submodule External/fmt updated: 0c9fce2ffe...873670ba3f.
Vendored
-1
Submodule External/imgui deleted from 4c986ecb8d.
Vendored
+1
-1
Submodule External/jemalloc updated: 7ae889695b...02ca52b5fe.
Vendored
+1
-1
Submodule External/jemalloc_glibc updated: 888181c5f7...404353974e.
Vendored
+1
-1
Submodule External/xbyak updated: f17cb9d6b9...c68cc53d18.
@@ -401,7 +401,7 @@ def print_parse_argloader_options(options):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "("
|
||||
conversion_func = "std::move("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
@@ -44,7 +44,7 @@ class OpDefinition:
|
||||
HasDest: bool
|
||||
DestType: str
|
||||
DestSize: str
|
||||
NumElements: str
|
||||
ElementSize: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
@@ -67,7 +67,7 @@ class OpDefinition:
|
||||
self.HasDest = False
|
||||
self.DestType = None
|
||||
self.DestSize = None
|
||||
self.NumElements = None
|
||||
self.ElementSize = None
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
@@ -101,7 +101,8 @@ def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR"):
|
||||
type == "FPR" or
|
||||
type == "PRED"):
|
||||
return True
|
||||
return False
|
||||
|
||||
@@ -150,8 +151,8 @@ def parse_ops(ops):
|
||||
RHS += f", {DType}:$Out{Name}"
|
||||
else:
|
||||
# Single anonymous destination
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR")
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR", "PRED"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR, PRED")
|
||||
|
||||
OpDef.HasDest = True
|
||||
OpDef.DestType = LHS
|
||||
@@ -221,7 +222,8 @@ def parse_ops(ops):
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpArg.Type == "FPR" or
|
||||
OpArg.Type == "PRED")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
@@ -232,8 +234,8 @@ def parse_ops(ops):
|
||||
if "DestSize" in op_val:
|
||||
OpDef.DestSize = op_val["DestSize"]
|
||||
|
||||
if "NumElements" in op_val:
|
||||
OpDef.NumElements = op_val["NumElements"]
|
||||
if "ElementSize" in op_val:
|
||||
OpDef.ElementSize = op_val["ElementSize"]
|
||||
|
||||
if len(op_class):
|
||||
OpDef.OpClass = op_class
|
||||
@@ -323,8 +325,8 @@ def print_ir_structs(defines):
|
||||
output_file.write("struct __attribute__((packed)) IROp_Header {\n")
|
||||
output_file.write("\tvoid* Data[0];\n")
|
||||
output_file.write("\tIROps Op;\n\n")
|
||||
output_file.write("\tuint8_t Size;\n")
|
||||
output_file.write("\tuint8_t ElementSize;\n")
|
||||
output_file.write("\tIR::OpSize Size;\n")
|
||||
output_file.write("\tIR::OpSize ElementSize;\n")
|
||||
|
||||
output_file.write("\ttemplate<typename T>\n")
|
||||
output_file.write("\tT const* C() const { return reinterpret_cast<T const*>(Data); }\n")
|
||||
@@ -630,20 +632,19 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElementSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetName(HeaderOp->Op));\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size / HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n")
|
||||
output_file.write("\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
|
||||
@@ -699,8 +700,12 @@ def print_ir_allocator_helpers():
|
||||
|
||||
# We gather the "has x87?" flag as we go. This saves the user from
|
||||
# having to keep track of whether they emitted any x87.
|
||||
# Also changes the mmx state to X87.
|
||||
if op.LoweredX87:
|
||||
output_file.write("\t\tRecordX87Use();\n")
|
||||
output_file.write(
|
||||
"\t\tif(MMXState == MMXState_MMX) ChgStateMMX_X87();\n"
|
||||
)
|
||||
|
||||
output_file.write("\t\tauto _Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
@@ -724,11 +729,11 @@ def print_ir_allocator_helpers():
|
||||
# We can only infer a size if we have arguments
|
||||
if op.DestSize == None:
|
||||
# We need to infer destination size
|
||||
output_file.write("\t\tuint8_t InferSize = 0;\n")
|
||||
output_file.write("\t\tIR::OpSize InferSize = OpSize::iUnsized;\n")
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tuint8_t Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
output_file.write("\t\tauto Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tInferSize = std::max(InferSize, Size{});\n".format(arg.Name))
|
||||
@@ -740,10 +745,10 @@ def print_ir_allocator_helpers():
|
||||
if op.DestSize != None:
|
||||
output_file.write("\t\t_Op.first->Header.Size = {};\n".format(op.DestSize))
|
||||
|
||||
if op.NumElements == None:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(1))
|
||||
if op.ElementSize == None:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size;\n")
|
||||
else:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = {};\n".format(op.ElementSize))
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
@@ -826,4 +831,3 @@ print_ir_dispatcher_defs()
|
||||
print_ir_dispatcher_dispatch()
|
||||
|
||||
output_dispatch_file.close()
|
||||
|
||||
@@ -86,7 +86,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
@@ -106,20 +105,19 @@ set (SRCS
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/JIT/Arm64/JIT.cpp
|
||||
Interface/Core/JIT/Arm64/ALUOps.cpp
|
||||
Interface/Core/JIT/Arm64/AtomicOps.cpp
|
||||
Interface/Core/JIT/Arm64/BranchOps.cpp
|
||||
Interface/Core/JIT/Arm64/ConversionOps.cpp
|
||||
Interface/Core/JIT/Arm64/EncryptionOps.cpp
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
Interface/Core/JIT/JIT.cpp
|
||||
Interface/Core/JIT/ALUOps.cpp
|
||||
Interface/Core/JIT/AtomicOps.cpp
|
||||
Interface/Core/JIT/BranchOps.cpp
|
||||
Interface/Core/JIT/ConversionOps.cpp
|
||||
Interface/Core/JIT/EncryptionOps.cpp
|
||||
Interface/Core/JIT/MemoryOps.cpp
|
||||
Interface/Core/JIT/MiscOps.cpp
|
||||
Interface/Core/JIT/MoveOps.cpp
|
||||
Interface/Core/JIT/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64Relocations.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
Interface/Core/X86Tables/H0F38Tables.cpp
|
||||
Interface/Core/X86Tables/H0F3ATables.cpp
|
||||
Interface/Core/X86Tables/PrimaryGroupTables.cpp
|
||||
@@ -128,8 +126,6 @@ set (SRCS
|
||||
Interface/Core/X86Tables/SecondaryTables.cpp
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
@@ -140,7 +136,6 @@ set (SRCS
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
Utils/Telemetry.cpp
|
||||
@@ -197,14 +192,6 @@ else()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
@@ -379,3 +366,25 @@ if (NOT MINGW_BUILD)
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
|
||||
# Meta-library to link jemalloc libraries enabled in the build configuration.
|
||||
# Only needed for targets that run emulation. For others, use JemallocDummy.
|
||||
add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
|
||||
if (ENABLE_JEMALLOC)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_JEMALLOC=1 JEMALLOC_NO_RENAME=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC FEX_jemalloc)
|
||||
endif()
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
|
||||
target_link_libraries(JemallocLibs INTERFACE FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
# Dummy project to use for host tools.
|
||||
# This overrides use of jemalloc in FEXCore with the normal glibc allocator.
|
||||
add_library(JemallocDummy STATIC Utils/AllocatorHooks.cpp)
|
||||
target_include_directories(JemallocDummy PRIVATE "${PROJECT_SOURCE_DIR}/include/")
|
||||
endif()
|
||||
|
||||
# The shared library should always link enabled jemalloc libraries
|
||||
target_link_libraries(${PROJECT_NAME}_shared JemallocLibs)
|
||||
@@ -55,7 +55,7 @@ void JITSymbols::RegisterJITSpace(const void* HostAddr, uint32_t CodeSize) {
|
||||
}
|
||||
|
||||
// Buffered JIT symbols.
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
void JITSymbols::Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -79,7 +79,7 @@ void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, u
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
void JITSymbols::Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -104,7 +104,7 @@ void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, u
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
void JITSymbols::RegisterNamedRegion(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -128,7 +128,7 @@ void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void*
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
void JITSymbols::WriteBuffer(FEXCore::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
auto Now = std::chrono::steady_clock::now();
|
||||
if (!ForceWrite) {
|
||||
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) && Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
|
||||
|
||||
@@ -11,6 +11,26 @@
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
constexpr static std::chrono::milliseconds MAXIMUM_THRESHOLD {100};
|
||||
|
||||
JITSymbolBuffer()
|
||||
: LastWrite {std::chrono::steady_clock::now()} {}
|
||||
// stead_clock to ensure a monotonic increasing clock.
|
||||
// In highly stressed situations this can still cause >2% CPU time in vdso_clock_gettime.
|
||||
// If we need lower CPU time when JIT symbols are enabled then FEX can read the cycle counter directly.
|
||||
std::chrono::steady_clock::time_point LastWrite {};
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
JITSymbols();
|
||||
@@ -21,16 +41,16 @@ public:
|
||||
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
|
||||
|
||||
// Allocate JIT buffer.
|
||||
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<Core::JITSymbolBuffer>();
|
||||
static fextl::unique_ptr<FEXCore::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<FEXCore::JITSymbolBuffer>();
|
||||
}
|
||||
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
|
||||
private:
|
||||
int fd {-1};
|
||||
void WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
void WriteBuffer(FEXCore::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -160,7 +160,31 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(state, lhs, rhs);
|
||||
/*
|
||||
* FPREM is not an IEEE-754 remainder. From the spec:
|
||||
*
|
||||
* Computes the remainder obtained from dividing the value in the ST(0)
|
||||
* register (the dividend) by the value in the ST(1) register (the divisor
|
||||
* or modulus), and stores the result in ST(0). The remainder represents the
|
||||
* following value:
|
||||
*
|
||||
* Remainder := ST(0) − (Q * ST(1))
|
||||
*
|
||||
* Here, Q is an integer value that is obtained by truncating the
|
||||
* floating-point number quotient of [ST(0) / ST(1)] toward zero.
|
||||
*
|
||||
* We implement this sequence literally. softfloat_round_minMag means
|
||||
* "truncate towards zero".
|
||||
*/
|
||||
extFloat80_t quotient = extF80_div(state, lhs, rhs);
|
||||
extFloat80_t Q = extF80_roundToInt(state, quotient, softfloat_round_minMag, true);
|
||||
bool Q_zero = Q.signif == 0 && (Q.signExp & ~(1 << 15)) == 0;
|
||||
|
||||
if (Q_zero) {
|
||||
return lhs;
|
||||
} else {
|
||||
return extF80_sub(state, lhs, extF80_mul(state, Q, rhs));
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -209,6 +233,10 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
// Zero is a special case, the significand for +/- 0 is +/- zero.
|
||||
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
@@ -232,6 +260,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
// Zero is a special case, the exponent is always -inf
|
||||
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
X80SoftFloat Result(1, 0x7FFFUL, 0x8000'0000'0000'0000UL);
|
||||
return Result;
|
||||
}
|
||||
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
@@ -262,6 +296,10 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
extFloat80_t Zero {0, 0};
|
||||
if (extF80_eq(state, lhs, Zero)) {
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Int = FRNDINT(state, rhs, softfloat_round_minMag);
|
||||
LIBRARY_PRECISION Src2_d = Int.ToFMax(state);
|
||||
Src2_d = exp2l(Src2_d);
|
||||
|
||||
@@ -43,7 +43,8 @@ namespace DefaultValues {
|
||||
} // namespace DefaultValues
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
PATH_DATA_DIR_GLOBAL,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
@@ -53,8 +54,8 @@ enum Paths {
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
void SetDataDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_DATA_DIR_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
@@ -73,15 +74,15 @@ const fextl::string& GetTelemetryDirectory() {
|
||||
Path = TelemetryDirectory;
|
||||
Path += "/";
|
||||
} else {
|
||||
Path = Config::GetDataDirectory() + "Telemetry/";
|
||||
Path = Config::GetDataDirectory(false) + "Telemetry/";
|
||||
}
|
||||
}
|
||||
|
||||
return Path;
|
||||
}
|
||||
|
||||
const fextl::string& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
const fextl::string& GetDataDirectory(bool Global) {
|
||||
return Paths[PATH_DATA_DIR_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetConfigDirectory(bool Global) {
|
||||
@@ -230,27 +231,26 @@ void Load() {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, fextl::string PathName) {
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
|
||||
// Expand home if it exists
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
PathName.replace(0, 1, Home);
|
||||
return PathName;
|
||||
if (PathName.starts_with("~/")) {
|
||||
Home.append(PathName.begin() + 1, PathName.end());
|
||||
return Home;
|
||||
}
|
||||
|
||||
// Expand relative path to absolute
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char* RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
if (RealPath && FHU::Filesystem::Exists(RealPath)) {
|
||||
return RealPath;
|
||||
}
|
||||
|
||||
// Only return if it exists
|
||||
@@ -318,81 +318,61 @@ fextl::string FindContainerPrefix() {
|
||||
void ReloadMetaLayer() {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 0;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
}
|
||||
|
||||
fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
const fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, const fextl::string& PathName) {
|
||||
const auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_ROOTFS);
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
fextl::string NamedRootFS = GetDataDirectory(false) + "RootFS/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, *PathName);
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, *PathName);
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKCONFIG);
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
fextl::string NamedConfig = GetDataDirectory(false) + "ThunkConfigs/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
|
||||
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
|
||||
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_OUTPUTLOG);
|
||||
if (*PathName != "stdout" && *PathName != "stderr" && *PathName != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, *PathName);
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) && !FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
FEX_CONFIG_OPT(PathName, DUMPIR);
|
||||
if (PathName() != "no") {
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_DUMPIR);
|
||||
if (*PathName != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
|
||||
@@ -1,19 +1,6 @@
|
||||
{
|
||||
"Options": {
|
||||
"CPU": {
|
||||
"Core": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
|
||||
"TextDefault": "irjit",
|
||||
"ShortArg": "c",
|
||||
"Choices": [ "irjit", "host" ],
|
||||
"ArgumentHandler": "CoreHandler",
|
||||
"Desc": [
|
||||
"Which CPU core to use",
|
||||
"host only exists on x86_64",
|
||||
"[irjit, host]"
|
||||
]
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
@@ -23,14 +24,6 @@ fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNe
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
|
||||
return CustomExitHandler;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
@@ -39,10 +32,6 @@ void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThrea
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator* _SignalDelegation) {
|
||||
SignalDelegation = _SignalDelegation;
|
||||
}
|
||||
@@ -52,6 +41,10 @@ void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandl
|
||||
SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetThunkHandler(FEXCore::ThunkHandler* Handler) {
|
||||
ThunkHandler = Handler;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
@@ -26,14 +26,8 @@
|
||||
#include <stdint.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <queue>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
@@ -65,16 +59,20 @@ namespace Validation {
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
enum CoreRunningMode {
|
||||
MODE_RUN = 0,
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct ExitFunctionLinkData {
|
||||
uint64_t HostBranch;
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
struct CustomIRResult {
|
||||
void* Creator;
|
||||
void* Data;
|
||||
|
||||
CustomIRResult(void* Creator, void* Data)
|
||||
: Creator(Creator)
|
||||
, Data(Data) {}
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
@@ -83,22 +81,18 @@ public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
ExitHandler GetExitHandler() const override;
|
||||
|
||||
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
|
||||
|
||||
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
|
||||
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
|
||||
|
||||
void ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) override;
|
||||
@@ -117,33 +111,29 @@ public:
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* - CTX->ExecuteThread(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
|
||||
* - ThreadHandler calls `CTX->ExecuteThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* - ExecuteThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) override;
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
@@ -151,6 +141,7 @@ public:
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator* SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) override;
|
||||
void SetThunkHandler(FEXCore::ThunkHandler* Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
@@ -190,9 +181,10 @@ public:
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
std::optional<CustomIRResult>
|
||||
AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
|
||||
void AppendThunkDefinitions(std::span<const FEXCore::IR::ThunkDefinition> Definitions) override;
|
||||
void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
@@ -203,7 +195,6 @@ public:
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
|
||||
@@ -223,12 +214,8 @@ public:
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
@@ -242,8 +229,6 @@ public:
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
} Config;
|
||||
|
||||
std::atomic_bool CoreShuttingDown {false};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
@@ -253,16 +238,9 @@ public:
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
@@ -313,19 +291,10 @@ public:
|
||||
[[nodiscard]]
|
||||
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState* ParentThread, FEXCore::Core::InternalThreadState* ChildThread);
|
||||
|
||||
uint8_t GetGPRSize() const {
|
||||
return Config.Is64BitMode ? 8 : 4;
|
||||
IR::OpSize GetGPROpSize() const {
|
||||
return Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
}
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
@@ -393,7 +362,6 @@ private:
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -58,6 +57,12 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r24, ARMEmitter::Reg::r25, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
|
||||
};
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// All are caller saved
|
||||
@@ -104,6 +109,12 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
@@ -235,6 +246,12 @@ namespace x32 {
|
||||
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
@@ -358,6 +375,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
GeneralRegisters = x64::RA;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PredicateRegisters = x64::PR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
@@ -371,6 +389,8 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
|
||||
PredicateRegisters = x32::PR;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -76,6 +76,9 @@ constexpr size_t CPU_AREA_EMULATOR_STACK_BASE_OFFSET = 0x8;
|
||||
constexpr size_t CPU_AREA_EMULATOR_DATA_OFFSET = 0x30;
|
||||
#endif
|
||||
|
||||
// Will force one single instruction block to be generated first if set when entering the JIT filling SRA.
|
||||
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP1;
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
@@ -94,6 +97,7 @@ protected:
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::PRegister> PredicateRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
@@ -1,51 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore {
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
std::fstream Output;
|
||||
Output.open("output.csv", std::fstream::out | std::fstream::binary);
|
||||
|
||||
if (!Output.is_open()) {
|
||||
return;
|
||||
}
|
||||
|
||||
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
|
||||
|
||||
for (auto it : SamplingMap) {
|
||||
if (!it.second->TotalCalls) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Output << "0x" << std::hex << it.first << ", " << std::dec << it.second->Min << ", " << std::dec << it.second->Max << ", " << std::dec
|
||||
<< it.second->TotalTime << ", " << std::dec << it.second->TotalCalls << ", " << std::dec
|
||||
<< ((double)it.second->TotalTime / (double)it.second->TotalCalls) << std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData* BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
auto it = SamplingMap.find(RIP);
|
||||
if (it != SamplingMap.end()) {
|
||||
return it->second;
|
||||
}
|
||||
BlockData* NewData = new BlockData {};
|
||||
memset(NewData, 0, sizeof(BlockData));
|
||||
NewData->Min = ~0ULL;
|
||||
SamplingMap[RIP] = NewData;
|
||||
return NewData;
|
||||
}
|
||||
|
||||
BlockSamplingData::~BlockSamplingData() {
|
||||
DumpBlockData();
|
||||
for (auto it : SamplingMap) {
|
||||
delete it.second;
|
||||
}
|
||||
SamplingMap.clear();
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -1,25 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockSamplingData {
|
||||
public:
|
||||
struct BlockData {
|
||||
uint64_t Start, End;
|
||||
uint64_t Min, Max;
|
||||
uint64_t TotalTime;
|
||||
uint64_t TotalCalls;
|
||||
};
|
||||
|
||||
BlockData* GetBlockData(uint64_t RIP);
|
||||
~BlockSamplingData();
|
||||
|
||||
void DumpBlockData();
|
||||
|
||||
private:
|
||||
std::unordered_map<uint64_t, BlockData*> SamplingMap;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -39,6 +39,15 @@ namespace CPU {
|
||||
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
|
||||
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
|
||||
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32_UPPER
|
||||
{0x5F00'0000'5F00'0000ULL, 0x5F00'0000'5F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I64
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32_UPPER
|
||||
{0x43E0'0000'0000'0000ULL, 0x43E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I64
|
||||
{0x8000'0000'8000'0000ULL, 0x8000'0000'8000'0000ULL}, // NAMED_VECTOR_CVTMAX_I32
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_I64
|
||||
{0x0000'0000'0000'0000ULL, 0x0000'0000'0000'8000ULL}, // NAMED_VECTOR_F80_SIGN_MASK
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
|
||||
@@ -48,11 +48,6 @@ namespace CPU {
|
||||
CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
/**
|
||||
* @return The name of this backend
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual fextl::string GetName() = 0;
|
||||
|
||||
struct CompiledCode {
|
||||
// Where this code block begins.
|
||||
@@ -85,9 +80,16 @@ namespace CPU {
|
||||
struct JITCodeTail {
|
||||
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
|
||||
size_t Size;
|
||||
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// The length of the guest code for this block.
|
||||
size_t GuestSize;
|
||||
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
@@ -124,17 +126,17 @@ namespace CPU {
|
||||
*
|
||||
* This is a thread specific compilation unit since there is one CPUBackend per guest thread
|
||||
*
|
||||
* If NeedsOpDispatch is returning false then IR and DebugData may be null and the expectation is that the code will still compile
|
||||
* FEXCore::Core::ThreadState* is valid at the time of compilation.
|
||||
*
|
||||
* @param Size - The byte size of the guest code for this block
|
||||
* @param SingleInst - If this block represents a single guest instruction
|
||||
* @param IR - IR that maps to the IR for this RIP
|
||||
* @param DebugData - Debug data that is available for this IR indirectly
|
||||
* @param CheckTF - If EFLAGS.TF checks should be emitted at the start of the block
|
||||
*
|
||||
* @return Information about the compiled code block.
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) = 0;
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
@@ -149,26 +151,6 @@ namespace CPU {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
*
|
||||
* @return Currently unused
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual void* MapRegion(void* HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
|
||||
*
|
||||
* This is useful if the FEXCore Frontend hits an x86-64 instruction that isn't understood but can continue regardless
|
||||
*
|
||||
* This is useful for example, a VM based CPUbackend
|
||||
*
|
||||
* @return true if it needs the IR
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual bool NeedsOpDispatch() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
/**
|
||||
|
||||
@@ -90,26 +90,45 @@ namespace ProductNames {
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
|
||||
static uint32_t GetCPUID() {
|
||||
uint32_t GetCPUID_Syscall() {
|
||||
uint32_t CPU {};
|
||||
FHU::Syscalls::getcpu(&CPU, nullptr);
|
||||
return CPU;
|
||||
}
|
||||
|
||||
struct CPUFamily {
|
||||
uint32_t Stepping : 4;
|
||||
uint32_t Model : 4;
|
||||
uint32_t ExtendedModel : 4;
|
||||
uint32_t FamilyID : 4;
|
||||
uint32_t ExtendedFamilyID : 8;
|
||||
uint32_t ProcessorType : 4;
|
||||
};
|
||||
|
||||
constexpr static uint32_t GenerateFamily(const CPUFamily Family) {
|
||||
return Family.Stepping | (Family.Model << 4) | (Family.FamilyID << 8) | (Family.ProcessorType << 12) | (Family.ExtendedModel << 16) |
|
||||
(Family.ExtendedFamilyID << 20);
|
||||
}
|
||||
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = 0 | // Stepping
|
||||
(0xA << 4) | // Model
|
||||
(0xF << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(0 << 16) | // Extended model ID
|
||||
(1 << 20); // Extended family ID
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
.Stepping = 0,
|
||||
.Model = 0xA,
|
||||
.ExtendedModel = 0,
|
||||
.FamilyID = 0xF,
|
||||
.ExtendedFamilyID = 1,
|
||||
.ProcessorType = 0,
|
||||
});
|
||||
|
||||
#else
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = 0 | // Stepping
|
||||
(0x7 << 4) | // Model
|
||||
(0x6 << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(1 << 16) | // Extended model ID
|
||||
(0x0 << 20); // Extended family ID
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
.Stepping = 1,
|
||||
.Model = 6,
|
||||
.ExtendedModel = 0xA,
|
||||
.FamilyID = 6,
|
||||
.ExtendedFamilyID = 0,
|
||||
.ProcessorType = 0,
|
||||
});
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
@@ -119,6 +138,12 @@ uint32_t GetCycleCounterFrequency() {
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint32_t GetCPUID_TPIDRRO() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], TPIDRRO_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
PerCPUData.resize(Cores);
|
||||
|
||||
@@ -876,11 +901,11 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) con
|
||||
// Extended processor and feature bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) const {
|
||||
|
||||
// RDTSCP is disabled on WIN32/Wine because there is no sane way to query processor ID.
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 1;
|
||||
#else
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 0;
|
||||
// RDTSCP under WIN32 is only supported if CPUIndex is available in TPIDRRO.
|
||||
const uint32_t SUPPORTS_RDTSCP = SupportsCPUIndexInTPIDRRO;
|
||||
#endif
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
@@ -1194,12 +1219,20 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
}
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {
|
||||
: CTX {ctx}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO}
|
||||
, GetCPUID {GetCPUID_Syscall} {
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
SetupFeatures();
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
if (SupportsCPUIndexInTPIDRRO) {
|
||||
GetCPUID = GetCPUID_TPIDRRO;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -115,6 +115,7 @@ public:
|
||||
|
||||
private:
|
||||
const FEXCore::Context::ContextImpl* CTX;
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
@@ -510,5 +511,8 @@ private:
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
}};
|
||||
|
||||
using GetCPUIDPtr = uint32_t (*)();
|
||||
GetCPUIDPtr GetCPUID;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -9,18 +9,16 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -35,6 +33,7 @@ $end_info$
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
@@ -79,9 +78,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
: HostFeatures {Features}
|
||||
, CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
@@ -116,13 +112,38 @@ ContextImpl::~ContextImpl() {
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
struct GetFrameBlockInfoResult {
|
||||
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
|
||||
const CPU::CPUBackend::JITCodeTail* InlineTail;
|
||||
};
|
||||
static GetFrameBlockInfoResult GetFrameBlockInfo(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader*>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
return {InlineHeader, InlineTail};
|
||||
}
|
||||
|
||||
return {InlineHeader, nullptr};
|
||||
}
|
||||
|
||||
bool ContextImpl::IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail && (Address + Size > InlineTail->RIP && Address < InlineTail->RIP + InlineTail->GuestSize);
|
||||
}
|
||||
|
||||
bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail && InlineTail->SingleInst;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto [InlineHeader, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
|
||||
if (InlineHeader) {
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries*>(
|
||||
Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
@@ -154,7 +175,8 @@ uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* T
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) {
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs,
|
||||
uint64_t PSTATE) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
uint32_t EFLAGS {};
|
||||
|
||||
@@ -164,6 +186,7 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
case X86State::RFLAG_TF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
@@ -216,6 +239,9 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
uint8_t TFByte = Frame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
EFLAGS |= (TFByte & 1) << X86State::RFLAG_TF_RAW_LOC;
|
||||
|
||||
// DF is pretransformed, undo the transform from 1/-1 back to 0/1
|
||||
uint8_t DFByte = Frame->State.flags[X86State::RFLAG_DF_RAW_LOC];
|
||||
if (DFByte & 0x80) {
|
||||
@@ -322,8 +348,6 @@ bool ContextImpl::InitCore() {
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.SupportsAVX = HostFeatures.SupportsAVX,
|
||||
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
|
||||
@@ -352,7 +376,6 @@ bool ContextImpl::InitCore() {
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#elif !defined(_M_ARM64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
@@ -361,8 +384,6 @@ bool ContextImpl::InitCore() {
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
// FEX needs to start paused when gdb is enabled.
|
||||
StartPaused = true;
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -372,32 +393,17 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason ContextImpl::RunUntilExit(FEXCore::Core::InternalThreadState* Thread) {
|
||||
ExecutionThread(Thread);
|
||||
|
||||
CoreShuttingDown.store(true);
|
||||
|
||||
if (CustomExitHandler) {
|
||||
CustomExitHandler(Thread, FEXCore::Context::ExitReason::EXIT_SHUTDOWN);
|
||||
return Thread->ExitReason;
|
||||
}
|
||||
|
||||
return FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
}
|
||||
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
}
|
||||
|
||||
|
||||
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
@@ -412,29 +418,23 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
|
||||
Thread->CTX = this;
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM: Thread->CPUBackend = CustomCPUFactory(this, Thread); break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown core configuration"); break;
|
||||
}
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {};
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
@@ -459,13 +459,7 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::C
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) {
|
||||
if (NeedsTLSUninstall) {
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCore::Allocator::VirtualProtect(&Thread->InterruptFaultPage, sizeof(Thread->InterruptFaultPage),
|
||||
Allocator::ProtectOptions::Read | Allocator::ProtectOptions::Write);
|
||||
delete Thread;
|
||||
@@ -505,7 +499,7 @@ void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, ui
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
@@ -586,6 +580,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst,
|
||||
[Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
@@ -599,7 +594,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount);
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
@@ -615,7 +610,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
@@ -658,8 +653,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -684,16 +678,23 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
++TotalInstructions;
|
||||
}
|
||||
} else {
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
if (!BlockInstructionsLength) {
|
||||
// SMC can modify block contents and patch invalid instructions to valid ones inline.
|
||||
// End blocks upon encountering them and only emit an invalid opcode exception if there are no prior instructions in the block (that could have modified it to be valid).
|
||||
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
}
|
||||
|
||||
HadInvalidInst = true;
|
||||
}
|
||||
|
||||
const bool NeedsBlockEnd =
|
||||
(HadDispatchError && TotalInstructions > 0) || (Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock);
|
||||
const bool NeedsBlockEnd = (HadDispatchError && TotalInstructions > 0) ||
|
||||
(Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock) || HadInvalidInst;
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
@@ -703,11 +704,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -782,6 +780,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t TotalInstructions {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
@@ -799,11 +798,12 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
|
||||
if (!IR) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
auto [IRCopy, _TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IR = std::move(IRCopy);
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
TotalInstructions = _TotalInstructions;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
}
|
||||
@@ -811,13 +811,17 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (!IR) {
|
||||
return {};
|
||||
}
|
||||
|
||||
// If the trap flag is set we generate single instruction blocks that each check to generate a single step exception.
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
auto IRView = IR->GetIRView();
|
||||
return {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, &IRView, DebugData, IR->RAData()).BlockEntry,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &IRView, DebugData, IR->RAData(), TFSet).BlockEntry,
|
||||
.IR = std::move(IR),
|
||||
.DebugData = DebugData,
|
||||
.GeneratedIR = true,
|
||||
@@ -902,43 +906,22 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileSingleStep");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Now notify the thread that we are initialized
|
||||
Thread->ThreadWaiting.NotifyAll();
|
||||
|
||||
if (StartPaused || Thread->StartPaused) {
|
||||
// Parent thread doesn't need to wait to run
|
||||
Thread->StartRunning.Wait();
|
||||
auto [CodePtr, IR, DebugData, GeneratedIR, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (!Thread->RunningEvents.EarlyExit.load()) {
|
||||
Thread->RunningEvents.WaitingToStart = false;
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE;
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
Thread->RunningEvents.Running = false;
|
||||
}
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
@@ -966,6 +949,10 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
if (!Thread) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
@@ -994,7 +981,8 @@ void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
std::optional<CustomIRResult>
|
||||
ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
@@ -1004,10 +992,51 @@ CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIR
|
||||
|
||||
if (!InsertedIterator.second) {
|
||||
const auto& [fn, Creator, Data] = InsertedIterator.first->second;
|
||||
return CustomIRResult(std::move(lk), Creator, Data);
|
||||
} else {
|
||||
lk.unlock();
|
||||
return CustomIRResult(std::move(lk), 0, 0);
|
||||
return CustomIRResult(Creator, Data);
|
||||
}
|
||||
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) {
|
||||
LOGMAN_THROW_AA_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
if (!Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}", Entrypoint, GuestThunkEntrypoint);
|
||||
|
||||
auto Result = AddCustomIREntrypoint(
|
||||
Entrypoint,
|
||||
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), X86State::REG_R11, IR::GPRClass, GPRSize);
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->_Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
ThunkHandler, (void*)GuestThunkEntrypoint);
|
||||
|
||||
if (Result.has_value()) {
|
||||
if (Result->Creator != ThunkHandler) {
|
||||
ERROR_AND_DIE_FMT("Input address for AddThunkTrampoline is already linked by another module");
|
||||
}
|
||||
if (Result->Data != (void*)GuestThunkEntrypoint) {
|
||||
// NOTE: This may happen in Vulkan thunks if the Vulkan driver resolves two different symbols
|
||||
// to the same function (e.g. vkGetPhysicalDeviceFeatures2/vkGetPhysicalDeviceFeatures2KHR)
|
||||
LogMan::Msg::EFmt("Input address for AddThunkTrampoline is already linked elsewhere");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1030,12 +1059,6 @@ void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(std::span<const FEXCore::IR::ThunkDefinition> Definitions) {
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
|
||||
@@ -46,6 +46,8 @@ Dispatcher::~Dispatcher() {
|
||||
}
|
||||
|
||||
void Dispatcher::EmitDispatcher() {
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
auto RipReg = TMP3;
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
#endif
|
||||
@@ -62,7 +64,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_CompileSingleStep;
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -81,6 +84,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
ARMEmitter::ForwardLabel CompileSingleStep;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
@@ -89,6 +93,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
|
||||
@@ -116,10 +124,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
auto RipReg = TMP3;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
@@ -204,37 +213,21 @@ void Dispatcher::EmitDispatcher() {
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Clobbers TMP1/2
|
||||
auto EmitSignalGuardedRegion = [&](auto Body) {
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
Body();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
@@ -250,15 +243,36 @@ void Dispatcher::EmitDispatcher() {
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
};
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
auto EmitECExitCheck = [&]() {
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::SingleUseForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
@@ -277,56 +291,83 @@ void Dispatcher::EmitDispatcher() {
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
};
|
||||
#endif
|
||||
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
// Jump to the compiled block
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
{
|
||||
Bind(&CompileSingleStep);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP1, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
b(&LoopTop);
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileSingleStep);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP }
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
// Jump to the compiled block
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -505,8 +546,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMF.GetConvertedPointer());
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMFCompileBlock.GetConvertedPointer());
|
||||
Bind(&l_CompileSingleStep);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
|
||||
dc64(PMFCompileSingleStep.GetConvertedPointer());
|
||||
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -220,7 +220,8 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
// DecodeInst->TableInfo may be null in the case of 3DNow! ModRM decoding.
|
||||
const bool IsIndexVector = DecodeInst->TableInfo && (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
uint8_t InvalidSIBIndex = 0b100; ///< SIB Index where there is no register encoding.
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
@@ -690,12 +691,8 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
}
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
|
||||
/* uint8_t P1 = */ ReadByte();
|
||||
/* uint8_t P2 = */ ReadByte();
|
||||
/* uint8_t P3 = */ ReadByte();
|
||||
uint8_t EVEXOp = ReadByte();
|
||||
return NormalOp(&EVEXTableOps[EVEXOp], EVEXOp);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
|
||||
@@ -930,7 +927,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
|
||||
uint64_t TargetRIP = 0;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
bool Conditional = true;
|
||||
|
||||
switch (DecodeInst->OP) {
|
||||
@@ -958,7 +955,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
default: return; break;
|
||||
}
|
||||
|
||||
if (GPRSize == 4) {
|
||||
if (GPRSize == IR::OpSize::i32Bit) {
|
||||
// If we are running a 32bit guest then wrap around addresses that go above 32bit
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
@@ -999,13 +996,13 @@ bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
}
|
||||
|
||||
uint64_t TargetRIP = 0;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
if (DecodeInst->OP == 0xE8) { // Call - immediate target
|
||||
const uint64_t NextRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
|
||||
if (GPRSize == 4) {
|
||||
if (GPRSize == IR::OpSize::i32Bit) {
|
||||
// If we are running a 32bit guest then wrap around addresses that go above 32bit
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
@@ -6,17 +6,20 @@
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW) {
|
||||
softfloat_state State;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
softfloat_state State {};
|
||||
State.detectTininess = softfloat_tininess_afterRounding;
|
||||
State.exceptionFlags = 0;
|
||||
State.roundingPrecision = 80;
|
||||
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
if (!Force80BitPrecision) {
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
}
|
||||
|
||||
auto RC = (FCW >> 10) & 3;
|
||||
@@ -132,7 +135,7 @@ struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FRNDINT(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -140,7 +143,7 @@ struct OpHandlers<IR::OP_F80ROUND> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::F2XM1(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -148,7 +151,7 @@ struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FTAN(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -164,7 +167,7 @@ struct OpHandlers<IR::OP_F80SQRT> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSIN(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -172,7 +175,7 @@ struct OpHandlers<IR::OP_F80SIN> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FCOS(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -226,7 +229,7 @@ struct OpHandlers<IR::OP_F80DIV> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FYL2X(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -234,7 +237,7 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FATAN(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -242,7 +245,7 @@ struct OpHandlers<IR::OP_F80ATAN> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM1(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -250,7 +253,7 @@ struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -258,7 +261,7 @@ struct OpHandlers<IR::OP_F80FPREM> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSCALE(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -322,8 +325,11 @@ struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
double trunc = (double)(int64_t)(src2); // truncate
|
||||
return src1 * exp2(trunc);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
}
|
||||
double trun = trunc(src2);
|
||||
return src1 * exp2(trun);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -79,17 +79,17 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
switch (IROp->Op) {
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -99,11 +99,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -115,7 +115,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
SupportsPreserveAllABI};
|
||||
@@ -124,7 +124,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
SupportsPreserveAllABI};
|
||||
@@ -133,7 +133,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
SupportsPreserveAllABI};
|
||||
@@ -156,11 +156,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
|
||||
+192
-128
@@ -5,9 +5,10 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -53,8 +54,8 @@ DEF_OP(EntrypointOffset) {
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg(Node);
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
const auto OpSize = IROp->Size;
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
@@ -91,10 +92,10 @@ DEF_OP(AddNZCV) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= 4, "Constant not allowed here");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmn(EmitSize, Src1, Const);
|
||||
} else if (IROp->Size < 4) {
|
||||
unsigned Shift = 32 - (8 * IROp->Size);
|
||||
} else if (IROp->Size < IR::OpSize::i32Bit) {
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
@@ -164,7 +165,7 @@ DEF_OP(TestNZ) {
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
// zeroing CV.
|
||||
if (IROp->Size < 4) {
|
||||
if (IROp->Size < IR::OpSize::i32Bit) {
|
||||
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
|
||||
// needed.
|
||||
if (Op->Src1 != Op->Src2) {
|
||||
@@ -178,7 +179,7 @@ DEF_OP(TestNZ) {
|
||||
Src1 = TMP1;
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (IROp->Size * 8);
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -190,6 +191,30 @@ DEF_OP(TestNZ) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TestZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestZ>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < IR::OpSize::i32Bit, "TestNZ used at higher sizes");
|
||||
const auto EmitSize = ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
uint64_t Mask = IROp->Size == IR::OpSize::i64Bit ? ~0ULL : ((1ull << IR::OpSizeAsBits(IROp->Size)) - 1);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
|
||||
LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
if (Src1 == Src2) {
|
||||
tst(EmitSize, Src1 /* Src2 */, Mask);
|
||||
} else {
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
tst(EmitSize, TMP1, Mask);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
@@ -198,25 +223,25 @@ DEF_OP(SubShift) {
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
|
||||
unsigned Shift = OpSize < IR::OpSize::i32Bit ? (32 - IR::OpSizeAsBits(OpSize)) : 0;
|
||||
ARMEmitter::Register ShiftedSrc1 = GetZeroableReg(Op->Src1);
|
||||
|
||||
// Shift to fix flags for <32-bit ops.
|
||||
// Any shift of zero is still zero so optimize out silly zero shifts.
|
||||
if (OpSize < 4 && ShiftedSrc1 != ARMEmitter::Reg::zr) {
|
||||
if (OpSize < IR::OpSize::i32Bit && ShiftedSrc1 != ARMEmitter::Reg::zr) {
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, ShiftedSrc1, Shift);
|
||||
ShiftedSrc1 = TMP1;
|
||||
}
|
||||
|
||||
if (OpSize < 4) {
|
||||
if (OpSize < IR::OpSize::i32Bit) {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()));
|
||||
@@ -261,10 +286,10 @@ DEF_OP(SetSmallNZV) {
|
||||
auto Op = IROp->C<IR::IROp_SetSmallNZV>();
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 1 || OpSize == 2, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
setf8(GetReg(Op->Src.ID()).W());
|
||||
} else {
|
||||
setf16(GetReg(Op->Src.ID()).W());
|
||||
@@ -272,8 +297,43 @@ DEF_OP(SetSmallNZV) {
|
||||
}
|
||||
|
||||
DEF_OP(AXFlag) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
|
||||
axflag();
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
axflag();
|
||||
} else {
|
||||
// AXFLAG is defined in the Arm spec as
|
||||
//
|
||||
// gt: nzCv -> nzCv
|
||||
// lt: Nzcv -> nzcv <==> 1 + 0
|
||||
// eq: nZCv -> nZCv <==> 1 + (~0)
|
||||
// un: nzCV -> nZcv <==> 0 + 0
|
||||
//
|
||||
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
|
||||
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
|
||||
auto V_inv = GetReg(IROp->Args[0].ID());
|
||||
csetm(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Condition::CC_EQ);
|
||||
ccmn(ARMEmitter::Size::i64Bit, V_inv, TMP1, ARMEmitter::StatusFlags {0x2} /* nzCv */, ARMEmitter::Condition::CC_LE);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Parity) {
|
||||
auto Op = IROp->C<IR::IROp_Parity>();
|
||||
auto Raw = GetReg(Op->Raw.ID());
|
||||
auto Dest = GetReg(Node);
|
||||
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, Raw, Raw, ARMEmitter::ShiftType::LSR, 4);
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 2);
|
||||
|
||||
if (Op->Invert) {
|
||||
eon(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
|
||||
} else {
|
||||
eor(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
|
||||
}
|
||||
|
||||
// The above sequence leaves garbage in the upper bits.
|
||||
if (Op->Mask) {
|
||||
and_(ARMEmitter::Size::i32Bit, Dest, Dest, 1);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
@@ -341,20 +401,20 @@ DEF_OP(Div) {
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
sxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
sxth(EmitSize, TMP1, Src1);
|
||||
sxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -370,20 +430,20 @@ DEF_OP(UDiv) {
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
uxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
uxth(EmitSize, TMP1, Src1);
|
||||
uxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -398,20 +458,20 @@ DEF_OP(Rem) {
|
||||
auto Op = IROp->C<IR::IROp_Rem>();
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
sxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
sxth(EmitSize, TMP1, Src1);
|
||||
sxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -427,20 +487,20 @@ DEF_OP(URem) {
|
||||
auto Op = IROp->C<IR::IROp_URem>();
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
uxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
uxth(EmitSize, TMP1, Src1);
|
||||
uxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -454,15 +514,15 @@ DEF_OP(URem) {
|
||||
|
||||
DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 4) {
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
sxtw(TMP1, Src1.W());
|
||||
sxtw(TMP2, Src2.W());
|
||||
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
|
||||
@@ -474,15 +534,15 @@ DEF_OP(MulH) {
|
||||
|
||||
DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 4) {
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
uxtw(ARMEmitter::Size::i64Bit, TMP1, Src1);
|
||||
uxtw(ARMEmitter::Size::i64Bit, TMP2, Src2);
|
||||
mul(ARMEmitter::Size::i64Bit, Dst, TMP1, TMP2);
|
||||
@@ -533,7 +593,7 @@ DEF_OP(Ornror) {
|
||||
|
||||
DEF_OP(AndWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AndWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
@@ -541,7 +601,7 @@ DEF_OP(AndWithFlags) {
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
// See TestNZ
|
||||
if (OpSize < 4) {
|
||||
if (OpSize < IR::OpSize::i32Bit) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
@@ -554,7 +614,7 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(OpSize);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Dst, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -580,7 +640,7 @@ DEF_OP(XornShift) {
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
auto Op = IROp->C<IR::IROp_Ashr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -588,29 +648,29 @@ DEF_OP(Ashr) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
asr(EmitSize, Dst, Src1, (unsigned int)Const);
|
||||
} else {
|
||||
sbfx(EmitSize, TMP1, Src1, 0, OpSize * 8);
|
||||
sbfx(EmitSize, TMP1, Src1, 0, IR::OpSizeAsBits(OpSize));
|
||||
asr(EmitSize, Dst, TMP1, (unsigned int)Const);
|
||||
ubfx(EmitSize, Dst, Dst, 0, OpSize * 8);
|
||||
ubfx(EmitSize, Dst, Dst, 0, IR::OpSizeAsBits(OpSize));
|
||||
}
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
asrv(EmitSize, Dst, Src1, Src2);
|
||||
} else {
|
||||
sbfx(EmitSize, TMP1, Src1, 0, OpSize * 8);
|
||||
sbfx(EmitSize, TMP1, Src1, 0, IR::OpSizeAsBits(OpSize));
|
||||
asrv(EmitSize, Dst, TMP1, Src2);
|
||||
ubfx(EmitSize, Dst, Dst, 0, OpSize * 8);
|
||||
ubfx(EmitSize, Dst, Dst, 0, IR::OpSizeAsBits(OpSize));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ShiftFlags) {
|
||||
auto Op = IROp->C<IR::IROp_ShiftFlags>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = Op->Size;
|
||||
const auto EmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto PFOutput = GetReg(Node);
|
||||
const auto PFInput = GetReg(Op->PFInput.ID());
|
||||
@@ -630,16 +690,16 @@ DEF_OP(ShiftFlags) {
|
||||
|
||||
// We need to mask the source before comparing it. We don't just skip flag
|
||||
// updates for Src2=0 but anything that masks to zero.
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == 8 ? 0x3f : 0x1f);
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
ands(EmitSize, PFTemp, Dst, Dst);
|
||||
} else {
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
unsigned Shift = 32 - (IR::OpSizeToSize(OpSize) * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Dst, ARMEmitter::ShiftType::LSL, Shift);
|
||||
mov(ARMEmitter::Size::i64Bit, PFTemp, Dst);
|
||||
}
|
||||
@@ -649,12 +709,12 @@ DEF_OP(ShiftFlags) {
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
if (Op->Shift == IR::ShiftType::LSL) {
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
neg(EmitSize, CFWord, Src2);
|
||||
lsrv(EmitSize, CFWord, Src1, CFWord);
|
||||
} else {
|
||||
CFWord = Dst.X();
|
||||
CFBit = (OpSize * 8);
|
||||
CFBit = IR::OpSizeToSize(OpSize) * 8;
|
||||
}
|
||||
} else {
|
||||
sub(ARMEmitter::Size::i64Bit, CFWord, Src2, 1);
|
||||
@@ -677,7 +737,7 @@ DEF_OP(ShiftFlags) {
|
||||
rmif(CFWord, (CFBit - 1) % 64, (1 << 1) /* C */);
|
||||
|
||||
if (SetOF) {
|
||||
rmif(TMP3, OpSize * 8 - 1, (1 << 0) /* V */);
|
||||
rmif(TMP3, IR::OpSizeToSize(OpSize) * 8 - 1, (1 << 0) /* V */);
|
||||
}
|
||||
} else {
|
||||
mrs(TMP2, ARMEmitter::SystemRegister::NZCV);
|
||||
@@ -690,7 +750,7 @@ DEF_OP(ShiftFlags) {
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP2, CFWord, 29 /* C */, 1);
|
||||
|
||||
if (SetOF) {
|
||||
lsr(EmitSize, TMP3, TMP3, OpSize * 8 - 1);
|
||||
lsr(EmitSize, TMP3, TMP3, IR::OpSizeToSize(OpSize) * 8 - 1);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP2, TMP3, 28 /* V */, 1);
|
||||
}
|
||||
|
||||
@@ -710,14 +770,14 @@ DEF_OP(RotateFlags) {
|
||||
const auto Result = GetReg(Op->Result.ID());
|
||||
const auto Shift = GetReg(Op->Shift.ID());
|
||||
const bool Left = Op->Left;
|
||||
const auto EmitSize = Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = Op->Size * 8;
|
||||
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
|
||||
unsigned CFBit = Left ? 0 : BitSize - 1;
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
@@ -837,7 +897,7 @@ DEF_OP(PDep) {
|
||||
DEF_OP(PExt) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto OpSizeBitsM1 = (OpSize * 8) - 1;
|
||||
const auto OpSizeBitsM1 = IR::OpSizeAsBits(OpSize) - 1;
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Input = GetReg(Op->Input.ID());
|
||||
@@ -892,8 +952,8 @@ DEF_OP(PExt) {
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -903,14 +963,14 @@ DEF_OP(LDiv) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
sxth(EmitSize, TMP2, Divisor);
|
||||
sdiv(EmitSize, Dst, TMP1, TMP2);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
@@ -918,7 +978,7 @@ DEF_OP(LDiv) {
|
||||
sdiv(EmitSize, Dst, TMP1, TMP2);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -962,8 +1022,8 @@ DEF_OP(LDiv) {
|
||||
|
||||
DEF_OP(LUDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -973,20 +1033,20 @@ DEF_OP(LUDiv) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64=
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
udiv(EmitSize, Dst, TMP1, Divisor);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
udiv(EmitSize, Dst, TMP1, Divisor);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1026,8 +1086,8 @@ DEF_OP(LUDiv) {
|
||||
|
||||
DEF_OP(LRem) {
|
||||
auto Op = IROp->C<IR::IROp_LRem>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -1037,7 +1097,7 @@ DEF_OP(LRem) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
sxth(EmitSize, TMP2, Divisor);
|
||||
@@ -1045,7 +1105,7 @@ DEF_OP(LRem) {
|
||||
msub(EmitSize, Dst, TMP3, TMP2, TMP1);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
@@ -1054,7 +1114,7 @@ DEF_OP(LRem) {
|
||||
msub(EmitSize, Dst, TMP2, TMP3, TMP1);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1100,8 +1160,8 @@ DEF_OP(LRem) {
|
||||
|
||||
DEF_OP(LURem) {
|
||||
auto Op = IROp->C<IR::IROp_LURem>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -1111,14 +1171,14 @@ DEF_OP(LURem) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
udiv(EmitSize, TMP2, TMP1, Divisor);
|
||||
msub(EmitSize, Dst, TMP2, Divisor, TMP1);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
@@ -1126,7 +1186,7 @@ DEF_OP(LURem) {
|
||||
msub(EmitSize, Dst, TMP2, Divisor, TMP1);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1178,30 +1238,30 @@ DEF_OP(Not) {
|
||||
|
||||
DEF_OP(Popcount) {
|
||||
auto Op = IROp->C<IR::IROp_Popcount>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 0x1:
|
||||
case IR::OpSize::i8Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
// only use lowest byte
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x2:
|
||||
case IR::OpSize::i16Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// only count two lowest bytes
|
||||
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x4:
|
||||
case IR::OpSize::i32Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// fmov has zero extended, unused bytes are zero
|
||||
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x8:
|
||||
case IR::OpSize::i64Bit:
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// fmov has zero extended, unused bytes are zero
|
||||
@@ -1220,34 +1280,27 @@ DEF_OP(FindLSB) {
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (IROp->Size != 8) {
|
||||
ubfx(EmitSize, TMP1, Src, 0, IROp->Size * 8);
|
||||
cmp(EmitSize, TMP1, 0);
|
||||
rbit(EmitSize, TMP1, TMP1);
|
||||
} else {
|
||||
rbit(EmitSize, TMP1, Src);
|
||||
cmp(EmitSize, Src, 0);
|
||||
}
|
||||
|
||||
// We assume the source is nonzero, so we can just rbit+clz without worrying
|
||||
// about upper garbage for smaller types.
|
||||
rbit(EmitSize, TMP1, Src);
|
||||
clz(EmitSize, Dst, TMP1);
|
||||
csinv(EmitSize, Dst, Dst, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_NE);
|
||||
}
|
||||
|
||||
DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
movz(ARMEmitter::Size::i64Bit, TMP1, OpSize * 8 - 1);
|
||||
movz(ARMEmitter::Size::i64Bit, TMP1, IR::OpSizeAsBits(OpSize) - 1);
|
||||
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
lsl(EmitSize, Dst, Src, 16);
|
||||
orr(EmitSize, Dst, Dst, 0x8000);
|
||||
clz(EmitSize, Dst, Dst);
|
||||
} else {
|
||||
clz(EmitSize, Dst, Src);
|
||||
@@ -1258,9 +1311,10 @@ DEF_OP(FindMSB) {
|
||||
|
||||
DEF_OP(FindTrailingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeroes>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1268,7 +1322,7 @@ DEF_OP(FindTrailingZeroes) {
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
// This orr does two things. First, if the (masked) source is zero, it
|
||||
// reverses to zero in the top so it forces clz to return 16. Second, it
|
||||
// ensures garbage in the upper bits of the source don't affect clz, because
|
||||
@@ -1282,15 +1336,16 @@ DEF_OP(FindTrailingZeroes) {
|
||||
|
||||
DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
// Expressing as lsl+orr+clz clears away any garbage in the upper bits
|
||||
// (alternatively could do uxth+clz+sub.. equal cost in total).
|
||||
lsl(EmitSize, Dst, Src, 16);
|
||||
@@ -1303,16 +1358,17 @@ DEF_OP(CountLeadingZeroes) {
|
||||
|
||||
DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
rev(EmitSize, Dst, Src);
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
lsr(EmitSize, Dst, Dst, 16);
|
||||
}
|
||||
}
|
||||
@@ -1338,10 +1394,10 @@ DEF_OP(Bfi) {
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (IROp->Size >= 4) {
|
||||
if (IROp->Size >= IR::OpSize::i32Bit) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, IROp->Size * 8);
|
||||
ubfx(EmitSize, Dst, TMP1, 0, IR::OpSizeAsBits(IROp->Size));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1372,7 +1428,7 @@ DEF_OP(Bfxil) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= IR::OpSize::i64Bit, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
@@ -1382,7 +1438,7 @@ DEF_OP(Bfe) {
|
||||
if (Op->lsb == 0 && Op->Width == 32) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
} else if (Op->lsb == 0 && Op->Width == 64) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == 8, "Must be 64-bit wide register");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == IR::OpSize::i64Bit, "Must be 64-bit wide register");
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
@@ -1399,9 +1455,9 @@ DEF_OP(Sbfe) {
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto CompareEmitSize = Op->CompareSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto CompareEmitSize = Op->CompareSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapCC(Op->Cond);
|
||||
@@ -1418,7 +1474,7 @@ DEF_OP(Select) {
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
fcmp(Op->CompareSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
fcmp(Op->CompareSize == IR::OpSize::i64Bit ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
|
||||
}
|
||||
@@ -1427,7 +1483,7 @@ DEF_OP(Select) {
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t all_ones = OpSize == IR::OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
@@ -1456,7 +1512,7 @@ DEF_OP(NZCVSelect) {
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t all_ones = IROp->Size == IR::OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
@@ -1475,6 +1531,14 @@ DEF_OP(NZCVSelect) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectV) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectV>();
|
||||
|
||||
auto cc = MapCC(Op->Cond);
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
fcsel(SubRegSize.Scalar, GetVReg(Node), GetVReg(Op->TrueVal.ID()), GetVReg(Op->FalseVal.ID()), cc);
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectIncrement) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectIncrement>();
|
||||
|
||||
@@ -1487,7 +1551,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = Op->Header.ElementSize * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
@@ -1498,10 +1562,10 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto PerformMove = [&](const ARMEmitter::VRegister reg, int index) {
|
||||
switch (OpSize) {
|
||||
case 1: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case 2: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case 4: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case 8: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i8Bit: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i16Bit: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i32Bit: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i64Bit: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize); break;
|
||||
}
|
||||
};
|
||||
@@ -1526,10 +1590,10 @@ DEF_OP(VExtractToGPR) {
|
||||
// upper half of the vector.
|
||||
const auto SanitizedIndex = [OpSize, Op] {
|
||||
switch (OpSize) {
|
||||
case 1: return Op->Index - 16;
|
||||
case 2: return Op->Index - 8;
|
||||
case 4: return Op->Index - 4;
|
||||
case 8: return Op->Index - 2;
|
||||
case IR::OpSize::i8Bit: return Op->Index - 16;
|
||||
case IR::OpSize::i16Bit: return Op->Index - 8;
|
||||
case IR::OpSize::i32Bit: return Op->Index - 4;
|
||||
case IR::OpSize::i64Bit: return Op->Index - 2;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OpSize: {}", OpSize); return 0;
|
||||
}
|
||||
}();
|
||||
@@ -1545,7 +1609,7 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
|
||||
if (Op->SrcElementSize == 8) {
|
||||
if (Op->SrcElementSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.D());
|
||||
} else {
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.S());
|
||||
@@ -1558,7 +1622,7 @@ DEF_OP(Float_ToGPR_S) {
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
|
||||
if (Op->SrcElementSize == 8) {
|
||||
if (Op->SrcElementSize == IR::OpSize::i64Bit) {
|
||||
frinti(VTMP1.D(), Src.D());
|
||||
fcvtzs(ConvertSize(IROp), Dst, VTMP1.D());
|
||||
} else {
|
||||
@@ -1569,7 +1633,7 @@ DEF_OP(Float_ToGPR_S) {
|
||||
|
||||
DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
const auto EmitSubSize = Op->ElementSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
const auto EmitSubSize = Op->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
|
||||
+3
-2
@@ -6,8 +6,9 @@ desc: relocation logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
+16
-13
@@ -7,13 +7,13 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
@@ -23,7 +23,7 @@ DEF_OP(CASPair) {
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = IROp->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
// RA has heuristics to try to pair sources, but we need to handle the cases
|
||||
// where they fail. We do so by moving to temporaries. Note we use 64-bit
|
||||
@@ -112,9 +112,9 @@ DEF_OP(CAS) {
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == 1) {
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
} else if (IROp->Size == 2) {
|
||||
} else if (IROp->Size == IR::OpSize::i16Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTH, 0);
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
@@ -273,18 +273,21 @@ DEF_OP(AtomicNeg) {
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
OpSize == IR::OpSize::i64Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i8Bit, "Unexpecte"
|
||||
"d CAS "
|
||||
"size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -294,7 +297,7 @@ DEF_OP(AtomicSwap) {
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
stlxr(SubEmitSize, TMP4, Src, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, OpSize * 8 - 1);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
|
||||
}
|
||||
}
|
||||
|
||||
+3
-3
@@ -9,13 +9,13 @@ $end_info$
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -117,7 +117,7 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
+35
-36
@@ -5,8 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -16,18 +15,18 @@ DEF_OP(VInsGPR) {
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(ElementSize);
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
@@ -91,16 +90,16 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 2:
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 4: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
|
||||
case 8: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
|
||||
case IR::OpSize::i32Bit: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
|
||||
case IR::OpSize::i64Bit: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
@@ -112,7 +111,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
@@ -127,8 +126,8 @@ DEF_OP(VDupFromGPR) {
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t ElementSize = IR::OpSizeToSize(Op->Header.ElementSize);
|
||||
const uint16_t Conv = (ElementSize << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
@@ -166,7 +165,7 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IR::OpSizeToSize(Op->Header.ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetVReg(Op->Scalar.ID());
|
||||
@@ -206,7 +205,7 @@ DEF_OP(Vector_SToF) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -216,15 +215,15 @@ DEF_OP(Vector_SToF) {
|
||||
scvtf(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
} else {
|
||||
if (OpSize == ElementSize) {
|
||||
if (ElementSize == 8) {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
|
||||
} else if (ElementSize == 4) {
|
||||
} else if (ElementSize == IR::OpSize::i32Bit) {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
|
||||
} else {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
|
||||
}
|
||||
} else {
|
||||
if (OpSize == 8) {
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
scvtf(SubEmitSize, Dst.D(), Vector.D());
|
||||
} else {
|
||||
scvtf(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
@@ -239,7 +238,7 @@ DEF_OP(Vector_FToZS) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -249,15 +248,15 @@ DEF_OP(Vector_FToZS) {
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
} else {
|
||||
if (OpSize == ElementSize) {
|
||||
if (ElementSize == 8) {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
|
||||
} else if (ElementSize == 4) {
|
||||
} else if (ElementSize == IR::OpSize::i32Bit) {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
|
||||
} else {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
|
||||
}
|
||||
} else {
|
||||
if (OpSize == 8) {
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(SubEmitSize, Dst.D(), Vector.D());
|
||||
} else {
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
@@ -270,7 +269,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
@@ -285,7 +284,7 @@ DEF_OP(Vector_FToS) {
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (OpSize == 8) {
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
} else {
|
||||
@@ -301,10 +300,10 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -404,7 +403,7 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -428,15 +427,15 @@ DEF_OP(Vector_FToI) {
|
||||
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
|
||||
// we can't just use a lambda without some seriously ugly casting.
|
||||
// This is fairly self-contained otherwise.
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == 2) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == 4) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == 8) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == IR::OpSize::i16Bit) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == IR::OpSize::i32Bit) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == IR::OpSize::i64Bit) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
@@ -465,7 +464,7 @@ DEF_OP(Vector_F64ToI32) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
+10
-10
@@ -5,7 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -24,7 +24,7 @@ DEF_OP(VAESEnc) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -49,7 +49,7 @@ DEF_OP(VAESEncLast) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -72,7 +72,7 @@ DEF_OP(VAESDec) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -97,7 +97,7 @@ DEF_OP(VAESDecLast) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -152,10 +152,10 @@ DEF_OP(CRC32) {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 2: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 4: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 8: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
|
||||
case IR::OpSize::i8Bit: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case IR::OpSize::i16Bit: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case IR::OpSize::i32Bit: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case IR::OpSize::i64Bit: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
@@ -193,7 +193,7 @@ DEF_OP(PCLMUL) {
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
+77
-23
@@ -16,7 +16,7 @@ $end_info$
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
@@ -484,11 +484,16 @@ static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
uintptr_t HostCode {};
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
if (!TFSet) {
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
}
|
||||
|
||||
if (!HostCode) {
|
||||
if (TFSet || !HostCode) {
|
||||
// If TF is set, the cache must be skipped as different code needs to be generated.
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
@@ -534,6 +539,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::PREDClass, PredicateRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
@@ -626,8 +632,8 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
const auto Size = OpHeader->Size;
|
||||
if (Size == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
@@ -654,8 +660,69 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::SingleUseForwardLabel l_TFUnset;
|
||||
ARMEmitter::SingleUseForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData,
|
||||
bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -711,22 +778,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
EmitInterruptChecks(CheckTF);
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -810,6 +862,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->GuestSize = Size;
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
{
|
||||
+48
-40
@@ -38,23 +38,9 @@ public:
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::string GetName() override {
|
||||
return "JIT";
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
|
||||
[[nodiscard]]
|
||||
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool NeedsOpDispatch() override {
|
||||
return true;
|
||||
}
|
||||
CPUBackend::CompiledCode
|
||||
CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
@@ -109,6 +95,19 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::PRegister GetPReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::PREDClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::PREDClass.Val) {
|
||||
return PredicateRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
@@ -144,23 +143,25 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == 4 || Op->Size == 8, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(uint8_t ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
return ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i128Bit;
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
return ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -169,8 +170,8 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(uint8_t ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != 16, "Invalid size");
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
@@ -181,13 +182,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 8, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
@@ -198,13 +199,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 16, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
@@ -241,7 +242,7 @@ private:
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
@@ -250,7 +251,7 @@ private:
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
ARMEmitter::SVEMemOperand GenerateSVEMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -331,20 +332,24 @@ private:
|
||||
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
void VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
void VFScalarFMAOperation(IR::OpSize OpSize, IR::OpSize ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Upper, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2,
|
||||
ARMEmitter::VRegister Addend);
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
void VFScalarOperation(IR::OpSize OpSize, IR::OpSize ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
void VFScalarUnaryOperation(IR::OpSize OpSize, IR::OpSize ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1,
|
||||
std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
void Emulate128BitGather(size_t Size, size_t ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, size_t VectorIndexSize,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
|
||||
void EmitInterruptChecks(bool CheckTF);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
@@ -367,4 +372,7 @@ private:
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,21 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+410
-302
File diff suppressed because it is too large.
Load diff
+12
-7
@@ -10,7 +10,7 @@ $end_info$
|
||||
#endif
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
@@ -192,8 +192,17 @@ DEF_OP(Print) {
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(ProcessorID) {
|
||||
if (CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::TPIDRRO_EL0);
|
||||
return;
|
||||
}
|
||||
#ifdef _WIN32
|
||||
else {
|
||||
// If on Windows and TPIDRRO isn't supported (like in wine), then this is a programming error.
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#else
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
|
||||
@@ -248,12 +257,8 @@ DEF_OP(ProcessorID) {
|
||||
// CPU is in w0
|
||||
// Node is in w1
|
||||
orr(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ShiftType::LSL, 12);
|
||||
}
|
||||
#else
|
||||
DEF_OP(ProcessorID) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
+1
-1
@@ -5,7 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
+335
-376
File diff suppressed because it is too large.
Load diff
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
@@ -34,7 +35,8 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// Allocate a region of memory that we can use to back our block pointers
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize));
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
// We need 32KB per guest page (One pointer per byte)
|
||||
@@ -52,8 +54,8 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
const size_t TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
ctx->SyscallHandler->UnmarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// No need to free BlockLinks map.
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
@@ -63,7 +65,7 @@ void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
@@ -71,7 +73,7 @@ void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,107 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable[] = {
|
||||
// Instructions
|
||||
{0x00, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0>},
|
||||
|
||||
{0x08, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0>},
|
||||
|
||||
{0x10, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0>},
|
||||
|
||||
{0x18, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0>},
|
||||
|
||||
{0x20, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0>},
|
||||
|
||||
{0x28, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0>},
|
||||
|
||||
{0x30, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0>},
|
||||
|
||||
{0x38, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0>},
|
||||
{0x50, 8, &OpDispatchBuilder::PUSHREGOp},
|
||||
{0x58, 8, &OpDispatchBuilder::POPOp},
|
||||
{0x68, 1, &OpDispatchBuilder::PUSHOp},
|
||||
{0x69, 1, &OpDispatchBuilder::IMUL2SrcOp},
|
||||
{0x6A, 1, &OpDispatchBuilder::PUSHOp},
|
||||
{0x6B, 1, &OpDispatchBuilder::IMUL2SrcOp},
|
||||
{0x6C, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{0x70, 16, &OpDispatchBuilder::CondJUMPOp},
|
||||
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0x86, 2, &OpDispatchBuilder::XCHGOp},
|
||||
{0x88, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
|
||||
{0x8C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, false>},
|
||||
{0x8D, 1, &OpDispatchBuilder::LEAOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, true>},
|
||||
{0x8F, 1, &OpDispatchBuilder::POPOp},
|
||||
{0x90, 8, &OpDispatchBuilder::XCHGOp},
|
||||
|
||||
{0x98, 1, &OpDispatchBuilder::CDQOp},
|
||||
{0x99, 1, &OpDispatchBuilder::CQOOp},
|
||||
{0x9B, 1, &OpDispatchBuilder::NOPOp},
|
||||
{0x9C, 1, &OpDispatchBuilder::PUSHFOp},
|
||||
{0x9D, 1, &OpDispatchBuilder::POPFOp},
|
||||
{0x9E, 1, &OpDispatchBuilder::SAHFOp},
|
||||
{0x9F, 1, &OpDispatchBuilder::LAHFOp},
|
||||
{0xA4, 2, &OpDispatchBuilder::MOVSOp},
|
||||
|
||||
{0xA6, 2, &OpDispatchBuilder::CMPSOp},
|
||||
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
{0xC2, 2, &OpDispatchBuilder::RETOp},
|
||||
{0xC8, 1, &OpDispatchBuilder::EnterOp},
|
||||
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
|
||||
{0xCC, 2, &OpDispatchBuilder::INTOp},
|
||||
{0xCF, 1, &OpDispatchBuilder::IRETOp},
|
||||
{0xD7, 2, &OpDispatchBuilder::XLATOp},
|
||||
{0xE0, 3, &OpDispatchBuilder::LoopOp},
|
||||
{0xE3, 1, &OpDispatchBuilder::CondJUMPRCXOp},
|
||||
{0xE4, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::CALLOp},
|
||||
{0xE9, 1, &OpDispatchBuilder::JUMPOp},
|
||||
{0xEB, 1, &OpDispatchBuilder::JUMPOp},
|
||||
{0xEC, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xF1, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xF4, 1, &OpDispatchBuilder::INTOp},
|
||||
|
||||
{0xF5, 1, &OpDispatchBuilder::FLAGControlOp},
|
||||
{0xF8, 2, &OpDispatchBuilder::FLAGControlOp},
|
||||
{0xFA, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xFC, 2, &OpDispatchBuilder::FLAGControlOp},
|
||||
};
|
||||
|
||||
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable_64[] = {
|
||||
{0x63, 1, &OpDispatchBuilder::MOVSXDOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
};
|
||||
|
||||
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable_32[] = {
|
||||
{0x06, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x07, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x0E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX>},
|
||||
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x17, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x1E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x1F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x27, 1, &OpDispatchBuilder::DAAOp},
|
||||
{0x2F, 1, &OpDispatchBuilder::DASOp},
|
||||
{0x37, 1, &OpDispatchBuilder::AAAOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::AASOp},
|
||||
{0x40, 8, &OpDispatchBuilder::INCOp},
|
||||
{0x48, 8, &OpDispatchBuilder::DECOp},
|
||||
|
||||
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
|
||||
{0x61, 1, &OpDispatchBuilder::POPAOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
{0xCE, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xD4, 1, &OpDispatchBuilder::AAMOp},
|
||||
{0xD5, 1, &OpDispatchBuilder::AADOp},
|
||||
{0xD6, 1, &OpDispatchBuilder::SALCOp},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -43,19 +43,19 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(16, 1, Dest, NewVec);
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -86,7 +86,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -121,25 +121,26 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto B = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto C = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto D = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
|
||||
@@ -147,13 +148,14 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(16, 4, Src, W_idx);
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
|
||||
@@ -165,12 +167,12 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(16, 4, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(16, 4, 0, Dest1, std::get<3>(Final));
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
|
||||
StoreResult(FPRClass, Op, Dest0, -1);
|
||||
StoreResult(FPRClass, Op, Dest0, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -183,52 +185,56 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
} else {
|
||||
const auto Sigma0 = [this](Ref W) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
return _Xor(
|
||||
OpSize::i32Bit,
|
||||
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 7)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 18))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 3)));
|
||||
};
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto W4 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto W3 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
Result = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, Sig1);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, Sig0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
const auto Sigma1 = [this](Ref W) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
return _Xor(
|
||||
OpSize::i32Bit,
|
||||
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 17)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 19))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 10)));
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), Sigma1(W17));
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W16);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
StoreResult(FPRClass, Op, D0, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
@@ -246,12 +252,12 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A,
|
||||
ShiftType::ROR, 22);
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E,
|
||||
ShiftType::ROR, 25);
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -259,64 +265,64 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
|
||||
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
|
||||
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
|
||||
auto Res3 = _VInsGPR(16, 4, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(16, 4, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(16, 4, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(16, 4, 0, Res1, E1);
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, -1);
|
||||
StoreResult(FPRClass, Op, Res0, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
@@ -325,19 +331,19 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
@@ -346,19 +352,19 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
@@ -367,19 +373,19 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
@@ -388,20 +394,20 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(16, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(16), RCON);
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(OpSize::i128Bit), RCON);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -409,19 +415,19 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,45 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_DDDTable[] = {
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
|
||||
{0xA0, 1, &OpDispatchBuilder::VPFCMPOp<2>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
|
||||
{0xB0, 1, &OpDispatchBuilder::VPFCMPOp<0>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB7, 1, &OpDispatchBuilder::PMULHRWOp},
|
||||
|
||||
{0xBB, 1, &OpDispatchBuilder::PSWAPDOp},
|
||||
{0xBF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -6,7 +6,6 @@ desc: Handles x86/64 flag generation
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
@@ -20,7 +19,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
FEXCore::X86State::RFLAG_CF_RAW_LOC, FEXCore::X86State::RFLAG_PF_RAW_LOC, FEXCore::X86State::RFLAG_AF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC, FEXCore::X86State::RFLAG_DF_RAW_LOC, FEXCore::X86State::RFLAG_OF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC, FEXCore::X86State::RFLAG_NT_LOC, FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC, FEXCore::X86State::RFLAG_AC_LOC, FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
@@ -37,13 +36,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
// Calculate flags early.
|
||||
// Could use InvalidateDeferredFlags() if we had masked invalidation.
|
||||
// This is only a partial overwrite of flags since OF isn't stored here.
|
||||
CalculateDeferredFlags();
|
||||
NumFlags = 5;
|
||||
} else {
|
||||
// We are overwriting all RFLAGS. Invalidate the deferred flag state.
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
// PF and CF are both stored inverted, so hoist the invert.
|
||||
@@ -77,16 +72,13 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Original = _Constant(0);
|
||||
|
||||
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_RAW_LOC)) && (FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
|
||||
// Handle CF first, since it's at bit 0 and hence doesn't need shift or OR.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_RAW_LOC == 0);
|
||||
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_RAW_LOC), "CF always handled");
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_RAW_LOC == 0);
|
||||
Ref Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
for (size_t i = 0; i < FlagOffsets.size(); ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
@@ -117,7 +109,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// instead.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
|
||||
// Set every bit except the bottommost.
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull));
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false, false), _InlineConstant(~1ull));
|
||||
|
||||
// Rotate the bottom bit to the appropriate location for PF, so we get
|
||||
// something like 111P1111. Then invert that to get 000p0000. Then OR that
|
||||
@@ -130,21 +122,21 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
if (GetNZ) {
|
||||
static_assert(FEXCore::X86State::RFLAG_SF_RAW_LOC == (FEXCore::X86State::RFLAG_ZF_RAW_LOC + 1));
|
||||
auto NZCV = GetNZCV();
|
||||
auto NZ = _And(OpSize::i64Bit, NZCV, _Constant(0b11u << 30));
|
||||
auto NZ = _And(OpSize::i64Bit, NZCV, _InlineConstant(0b11u << 30));
|
||||
Original = _Orlshr(OpSize::i64Bit, Original, NZ, 31 - FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
|
||||
// The constant is OR'ed in at the end, to avoid a pointless or xzr, #2.
|
||||
if ((1U << X86State::RFLAG_RESERVED_LOC) & FlagsMask) {
|
||||
Original = _Or(OpSize::i64Bit, Original, _Constant(2));
|
||||
Original = _Or(OpSize::i64Bit, Original, _InlineConstant(2));
|
||||
}
|
||||
|
||||
return Original;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t SignBit = (SrcSize * 8) - 1;
|
||||
void OpDispatchBuilder::CalculateOF(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const uint64_t SignBit = IR::OpSizeAsBits(SrcSize) - 1;
|
||||
Ref Anded = nullptr;
|
||||
|
||||
// For add, OF is set iff the sources have the same sign but the destination
|
||||
@@ -175,25 +167,15 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2
|
||||
}
|
||||
}
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SignBit, true);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadPFRaw(bool Invert) {
|
||||
// Read the stored byte. This is the original result (up to 64-bits), it needs
|
||||
// parity calculated.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
Ref OpDispatchBuilder::LoadPFRaw(bool Mask, bool Invert) {
|
||||
// Most blocks do not read parity, so PF optimization is gated on this flag.
|
||||
CurrentHeader->ReadsParity = true;
|
||||
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4);
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2);
|
||||
|
||||
if (Invert) {
|
||||
Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
|
||||
} else {
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
|
||||
}
|
||||
|
||||
return Result;
|
||||
// Evaluate parity on the deferred raw value.
|
||||
return _Parity(GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC), Mask, Invert);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadAF() {
|
||||
@@ -203,8 +185,9 @@ Ref OpDispatchBuilder::LoadAF() {
|
||||
// Read the result, stored for PF.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// What's left is to XOR and extract. This is the deferred part.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFWord, Result));
|
||||
// What's left is to XOR and extract. This is the deferred part. We
|
||||
// specifically use a 64-bit Xor here as we don't need masking.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i64Bit, AFWord, Result));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FixupAF() {
|
||||
@@ -217,7 +200,8 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
// Again 64-bit as masking is more expensive given our ConstProp design.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -256,8 +240,8 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
// appropriate bit. Again 64-bit to avoid masking.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -276,22 +260,22 @@ Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref Res;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
RectifyCarryInvert(false);
|
||||
HandleNZCV_RMW();
|
||||
Res = _AdcWithFlags(OpSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
} else {
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
@@ -299,7 +283,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
@@ -313,15 +297,15 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
// Arm's subtraction has inverted CF from x86, so rectify the input and
|
||||
// invert the output.
|
||||
RectifyCarryInvert(true);
|
||||
@@ -330,13 +314,13 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
CFInverted = true;
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, SrcSize * 8, 0, Src1);
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
Src1 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src1);
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
|
||||
@@ -349,7 +333,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(IR::OpSize SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
@@ -358,10 +342,10 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _SubWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _SubWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_SubNZCV(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
_SubNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Sub(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
@@ -379,7 +363,7 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
return Res;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
@@ -388,10 +372,10 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _AddWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _AddWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_AddNZCV(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
_AddNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Add(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
@@ -408,18 +392,18 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
return Res;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High) {
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High) {
|
||||
HandleNZCVWrite();
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, IR::OpSizeAsBits(SrcSize) - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
// undefined, this does what we need after inverting carry.
|
||||
auto Zero = _Constant(0);
|
||||
auto Zero = _InlineConstant(0);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
@@ -428,8 +412,8 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
HandleNZCVWrite();
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
|
||||
auto Zero = _InlineConstant(0);
|
||||
const auto Size = GetOpSize(High);
|
||||
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
@@ -441,7 +425,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
InvalidateAF();
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -450,13 +434,13 @@ void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, UnmaskedRes);
|
||||
|
||||
@@ -465,7 +449,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
// Extract the last bit shifted in to CF. Shift is already masked, but for
|
||||
// 8/16-bit it might be >= SrcSizeBits, in which case CF is cleared. There's
|
||||
// nothing to do in that case since we already cleared CF above.
|
||||
auto SrcSizeBits = SrcSize * 8;
|
||||
const auto SrcSizeBits = IR::OpSizeAsBits(SrcSize);
|
||||
if (Shift < SrcSizeBits) {
|
||||
SetCFDirect(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
@@ -478,13 +462,13 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
if (Shift == 1) {
|
||||
auto Xor = _Xor(OpSize, UnmaskedRes, Src1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, IR::OpSizeAsBits(SrcSize) - 1, true);
|
||||
} else {
|
||||
// Undefined, we choose to zero as part of SetNZ_ZeroCV
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -504,7 +488,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
// already zeroed there's nothing to do here.
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// Set SF and PF. Clobbers OF, but OF only defined for Shift = 1 where it is
|
||||
// set below.
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -516,7 +500,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
InvalidateAF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -529,18 +513,18 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// Is set to the MSB of the original value
|
||||
if (Shift == 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Src1, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Src1, IR::OpSizeAsBits(SrcSize) - 1, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
CalculateFlags_ShiftRightImmediateCommon(SrcSize, Res, Src1, Shift);
|
||||
|
||||
// OF
|
||||
@@ -550,12 +534,12 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
// XOR of Result and Src1
|
||||
if (Shift == 1) {
|
||||
auto val = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, IR::OpSizeAsBits(SrcSize) - 1, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(IR::OpSize SrcSize, Ref Result) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
// Test ZF of result, SF is undefined so this is ok.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
@@ -563,7 +547,7 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// Now set CF if the Result = SrcSize * 8. Since SrcSize is a power-of-two and
|
||||
// Result is <= SrcSize * 8, we equivalently check if the log2(SrcSize * 8)
|
||||
// bit is set. No masking is needed because no higher bits could be set.
|
||||
unsigned CarryBit = FEXCore::ilog2(SrcSize * 8u);
|
||||
unsigned CarryBit = FEXCore::ilog2(IR::OpSizeAsBits(SrcSize));
|
||||
SetCFDirect(Result, CarryBit);
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_66, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_NONE, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_66, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_NONE, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_66, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_66, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
|
||||
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x10), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x17), 1, &OpDispatchBuilder::PTestOp},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i8Bit>},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x3A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x3B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x3C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x3D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x3E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
{OPD(PF_38_66, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
{OPD(PF_38_F3, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,77 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
constexpr auto OpDispatchTableGenH0F3AREX = []<uint16_t REX>() consteval {
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> Table[] = {
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
|
||||
auto REX0 = OpDispatchTableGenH0F3AREX.template operator()<0>();
|
||||
auto REX1 = OpDispatchTableGenH0F3AREX.template operator()<1>();
|
||||
auto concat = []<typename T, size_t N1, size_t N2>(std::array<T, N1> const& lhs,
|
||||
std::array<T, N2> const& rhs) consteval -> std::array<T, N1 + N2> {
|
||||
std::array<T, N1 + N2> Table {};
|
||||
for (size_t i = 0; i < N1; ++i) {
|
||||
Table[i] = lhs[i];
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < N2; ++i) {
|
||||
Table[N1 + i] = rhs[i];
|
||||
}
|
||||
|
||||
return Table;
|
||||
};
|
||||
return concat(REX0, REX1);
|
||||
};
|
||||
|
||||
constexpr auto OpDispatch_H0F3ATableIgnoreREX = OpDispatchTableGenH0F3A();
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
|
||||
#undef OPD
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,129 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
using X86Tables::OpToIndex;
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_PrimaryGroupTables[] = {
|
||||
// GROUP 1
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>}, // CMP
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
|
||||
// GROUP 2
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, false>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, false>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 2), 1, &OpDispatchBuilder::RCLOp1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 3), 1, &OpDispatchBuilder::RCROp8x1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, true>}, // 1Bit SHR
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, true>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 2), 1, &OpDispatchBuilder::RCLOp1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 3), 1, &OpDispatchBuilder::RCROp1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, true>}, // 1Bit SHR
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, true>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::RCLSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::RCRSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
|
||||
|
||||
// GROUP 3
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, &OpDispatchBuilder::MULOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 5), 1, &OpDispatchBuilder::IMULOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 4), 1, &OpDispatchBuilder::MULOp}, // MUL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 5), 1, &OpDispatchBuilder::IMULOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
|
||||
|
||||
// GROUP 4
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_4, OpToIndex(0xFE), 0), 1, &OpDispatchBuilder::INCOp}, // INC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_4, OpToIndex(0xFE), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
|
||||
|
||||
// GROUP 5
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, &OpDispatchBuilder::INCOp}, // INC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, &OpDispatchBuilder::CALLAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, &OpDispatchBuilder::JUMPAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp},
|
||||
|
||||
// GROUP 11
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 1>},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,161 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
constexpr uint16_t PF_NONE = 0;
|
||||
constexpr uint16_t PF_F3 = 1;
|
||||
constexpr uint16_t PF_66 = 2;
|
||||
constexpr uint16_t PF_F2 = 3;
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryGroupTables[] = {
|
||||
// GROUP 6
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_F2, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 7), 1, &OpDispatchBuilder::RDPIDOp},
|
||||
|
||||
// GROUP 12
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i16Bit>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i16Bit>},
|
||||
|
||||
// GROUP 13
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i32Bit>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i32Bit>},
|
||||
|
||||
// GROUP 14
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i64Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i64Bit>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i64Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 3), 1, &OpDispatchBuilder::PSRLDQ},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i64Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLLDQ},
|
||||
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 0), 1, &OpDispatchBuilder::FXSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 1), 1, &OpDispatchBuilder::FXRStoreOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 2), 1, &OpDispatchBuilder::LDMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 3), 1, &OpDispatchBuilder::STMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 4), 1, &OpDispatchBuilder::XSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::LoadFenceOrXRSTOR}, // LFENCE (or XRSTOR)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, // MFENCE (or XSAVEOPT)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, // SFENCE (or CLFLUSH)
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 6), 1, &OpDispatchBuilder::CLWB},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 7), 1, &OpDispatchBuilder::CLFLUSHOPT},
|
||||
|
||||
// GROUP 16
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
// GROUP P
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 3), 5, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F3, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_66, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryGroupTables_64[] = {
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,22 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryModRMTables[] = {
|
||||
// REG /1
|
||||
{((0 << 3) | 0), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{((0 << 3) | 1), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
// REG /2
|
||||
{((1 << 3) | 0), 1, &OpDispatchBuilder::XGetBVOp},
|
||||
|
||||
// REG /3
|
||||
{((2 << 3) | 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 0), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,329 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x0B, 1, &OpDispatchBuilder::INTOp},
|
||||
{0x0E, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp}, // NOP with ModRM
|
||||
|
||||
{0x20, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{0x30, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x31, 1, &OpDispatchBuilder::RDTSCOp},
|
||||
{0x32, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x34, 3, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
{0x80, 16, &OpDispatchBuilder::CondJUMPOp},
|
||||
{0x90, 16, &OpDispatchBuilder::SETccOp},
|
||||
{0xA2, 1, &OpDispatchBuilder::CPUIDOp},
|
||||
{0xA3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTNone>}, // BT
|
||||
{0xA4, 1, &OpDispatchBuilder::SHLDImmediateOp},
|
||||
{0xA5, 1, &OpDispatchBuilder::SHLDOp},
|
||||
{0xAB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTSet>}, // BTS
|
||||
{0xAC, 1, &OpDispatchBuilder::SHRDImmediateOp},
|
||||
{0xAD, 1, &OpDispatchBuilder::SHRDOp},
|
||||
{0xAF, 1, &OpDispatchBuilder::IMUL1SrcOp},
|
||||
{0xB0, 2, &OpDispatchBuilder::CMPXCHGOp}, // CMPXCHG
|
||||
{0xB3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTClear>}, // BTR
|
||||
{0xB6, 2, &OpDispatchBuilder::MOVZXOp},
|
||||
{0xBB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTComplement>}, // BTC
|
||||
{0xBC, 1, &OpDispatchBuilder::BSFOp}, // BSF
|
||||
{0xBD, 1, &OpDispatchBuilder::BSROp}, // BSF
|
||||
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
|
||||
{0xC0, 2, &OpDispatchBuilder::XADDOp},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
|
||||
|
||||
// SSE
|
||||
{0x10, 2, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x12, 2, &OpDispatchBuilder::MOVLPOp},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x53, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i32Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFW8ByteOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i8Bit>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x77, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i32Bit>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i32Bit>},
|
||||
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i32Bit>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i64Bit>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i64Bit>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i16Bit>},
|
||||
{0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i8Bit>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i16Bit>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i8Bit>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i64Bit>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i8Bit>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i16Bit>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i8Bit>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i64Bit>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i8Bit>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i16Bit>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i16Bit>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i8Bit>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i16Bit>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i32Bit>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i64Bit>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i8Bit>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSSOp},
|
||||
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
|
||||
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i32Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, false>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
|
||||
{0xBC, 1, &OpDispatchBuilder::TZCNT},
|
||||
{0xBD, 1, &OpDispatchBuilder::LZCNT},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
|
||||
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i64Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
// x52 = Invalid
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i64Bit>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x12, 2, &OpDispatchBuilder::MOVLPOp},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i64Bit>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i64Bit>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i64Bit>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i64Bit>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
|
||||
{0x6C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i64Bit>},
|
||||
{0x6D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i64Bit>},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFDOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i8Bit>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i64Bit>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i64Bit>},
|
||||
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i64Bit>},
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i32Bit>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i64Bit>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i64Bit>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i16Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i8Bit>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i16Bit>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i8Bit>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i8Bit>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i16Bit>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i8Bit>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i8Bit>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i16Bit>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i16Bit>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i8Bit>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i16Bit>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i32Bit>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i64Bit>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i8Bit>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable_64[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SyscallOp, true>},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable_32[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::NOPOp},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,26 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(map_select, pp, opcode) (((map_select - 1) << 10) | (pp << 8) | (opcode))
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_VEXTable[] = {
|
||||
{OPD(2, 0b00, 0xF2), 1, &OpDispatchBuilder::ANDNBMIOp}, {OPD(2, 0b00, 0xF5), 1, &OpDispatchBuilder::BZHI},
|
||||
{OPD(2, 0b10, 0xF5), 1, &OpDispatchBuilder::PEXT}, {OPD(2, 0b11, 0xF5), 1, &OpDispatchBuilder::PDEP},
|
||||
{OPD(2, 0b11, 0xF6), 1, &OpDispatchBuilder::MULX}, {OPD(2, 0b00, 0xF7), 1, &OpDispatchBuilder::BEXTRBMIOp},
|
||||
{OPD(2, 0b01, 0xF7), 1, &OpDispatchBuilder::BMI2Shift}, {OPD(2, 0b10, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
{OPD(2, 0b11, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
|
||||
{OPD(3, 0b11, 0xF0), 1, &OpDispatchBuilder::RORX},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
#define OPD(group, pp, opcode) (((group - X86Tables::InstType::TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
constexpr std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_VEXGroupTable[] = {
|
||||
{OPD(X86Tables::InstType::TYPE_VEX_GROUP_17, 0, 0b001), 1, &OpDispatchBuilder::BLSRBMIOp},
|
||||
{OPD(X86Tables::InstType::TYPE_VEX_GROUP_17, 0, 0b010), 1, &OpDispatchBuilder::BLSMSKBMIOp},
|
||||
{OPD(X86Tables::InstType::TYPE_VEX_GROUP_17, 0, 0b011), 1, &OpDispatchBuilder::BLSIBMIOp},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
File diff suppressed because it is too large.
Load diff
@@ -26,7 +26,7 @@ class OrderedNode;
|
||||
Ref OpDispatchBuilder::GetX87Top() {
|
||||
// Yes, we are storing 3 bits in a single flag register.
|
||||
// Deal with it
|
||||
return _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
@@ -56,17 +56,17 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, size_t Width) {
|
||||
size_t ReadWidth = (Width == 80) ? 16 : Width / 8;
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref ConvertedData = Data;
|
||||
// Convert to 80bit float
|
||||
if (Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
ConvertedData = _F80CVTTo(Data, ReadWidth);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
@@ -79,31 +79,31 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
_PushStack(ConvertedData, Data, 16, true);
|
||||
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
Ref converted = _F80BCDStore(_ReadStackValue(0));
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant Constant) {
|
||||
// Update TOP
|
||||
Ref Data = LoadAndCacheNamedVectorConstant(16, Constant);
|
||||
_PushStack(Data, Data, 16, true);
|
||||
Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, Constant);
|
||||
_PushStack(Data, Data, OpSize::i128Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
size_t ReadWidth = GetSrcSize(Op);
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (ReadWidth != 8) {
|
||||
Data = _Sbfe(OpSize::i64Bit, ReadWidth * 8, 0, Data);
|
||||
if (ReadWidth != OpSize::i64Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
|
||||
// We're about to clobber flags to grab the sign, so save NZCV.
|
||||
@@ -123,14 +123,14 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VCastFromGPR(16, 8, shifted);
|
||||
ConvertedData = _VInsElement(16, 8, 1, 0, ConvertedData, _VCastFromGPR(16, 8, upper));
|
||||
Ref ConvertedData = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, shifted);
|
||||
ConvertedData = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, ConvertedData, _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, upper));
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, size_t Width) {
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i128Bit, true, Width / 8);
|
||||
_StoreStackMemory(Mem, OpSize::i128Bit, true, Width);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
@@ -149,20 +149,18 @@ void OpDispatchBuilder::FSTToStack(OpcodeArgs) {
|
||||
|
||||
// Store integer to memory (possibly with truncation)
|
||||
void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
// FIXME(pmatos): is there any advantage of using STORESTACKMEMORY here?
|
||||
// Do we need STORESTACKMEMORY at all?
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Data = _ReadStackValue(0);
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, 1);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FADD(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto Offset = Op->OP & 7;
|
||||
auto St0 = 0;
|
||||
@@ -177,23 +175,22 @@ void OpDispatchBuilder::FADD(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
_F80AddValue(0, Arg);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto offset = Op->OP & 7;
|
||||
auto st0 = 0;
|
||||
@@ -208,16 +205,15 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -228,7 +224,7 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
@@ -246,16 +242,15 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -270,7 +265,7 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSUB(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
@@ -288,16 +283,15 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -348,42 +342,42 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
// Before we store anything we need to sync our stack to the registers.
|
||||
_SyncStackToSlow();
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -406,26 +400,27 @@ Ref OpDispatchBuilder::ReconstructX87StateFromFSW_Helper(Ref FSW) {
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(IR::OpSizeToSize(Size) * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(IR::OpSizeToSize(Size) * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
|
||||
// 14 bytes for 16bit
|
||||
// 2 Bytes : FCW
|
||||
// 2 Bytes : FSW
|
||||
@@ -444,60 +439,66 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// 2 bytes : Opcode
|
||||
// 4 bytes : data pointer offset
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
auto data = _LoadContextIndexed(Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreMem(FPRClass, 16, data, Mem, _Constant((Size * 7) + (10 * i)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
auto data = _LoadContextIndexed(Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, data, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, topBytes, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
|
||||
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -505,17 +506,27 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (ReducedPrecisionMode) {
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = NewFCW;
|
||||
auto roundShift = _Constant(10);
|
||||
auto roundMask = _Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
}
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
@@ -523,15 +534,18 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
Ref Mask = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, low);
|
||||
Mask = _VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, Mask, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, 16, Mem, _Constant((Size * 7) + (10 * i)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
|
||||
_StoreContextIndexed(Reg, Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
@@ -540,29 +554,31 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg = _LoadMem(FPRClass, 8, Mem, _Constant((Size * 7) + (10 * 7)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, 2, Mem, _Constant((Size * 7) + (10 * 7) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
_StoreContextIndexed(Reg, Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh =
|
||||
_LoadMem(FPRClass, OpSize::i16Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
}
|
||||
|
||||
// Load / Store Control Word
|
||||
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, -1);
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
// FIXME: Because loading control flags will affect several instructions in fast path, we might have
|
||||
// to switch for now to slow mode whenever these are manually changed.
|
||||
// Remove the next line and try DF_04.asm in fast path.
|
||||
_StackForceSlow();
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
uint8_t Offset = Op->OP & 7;
|
||||
// fxch st0, st0 is for us essentially a nop
|
||||
@@ -575,15 +591,15 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87FYL2X(OpcodeArgs, bool IsFYL2XP1) {
|
||||
if (IsFYL2XP1) {
|
||||
// create an add between top of stack and 1.
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000)) :
|
||||
LoadAndCacheNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, _Constant(0x3FF0000000000000)) :
|
||||
LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
_F80AddValue(0, One);
|
||||
}
|
||||
|
||||
_F80FYL2XStack();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
@@ -593,15 +609,17 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatch
|
||||
uint8_t Offset = Op->OP & 7;
|
||||
Res = _F80CmpStack(Offset);
|
||||
} else {
|
||||
// Memory arg
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width / 8);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width / 8);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
Res = _F80CmpValue(b);
|
||||
}
|
||||
@@ -618,10 +636,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatch
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
} else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetCFDirect(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
@@ -681,7 +696,6 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
|
||||
// Optionally we can pass a pre calculated value for Top, otherwise we calculate it
|
||||
// during the function runtime.
|
||||
Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
|
||||
// Start with the top value
|
||||
auto Top = T ? T : GetX87Top();
|
||||
Ref FSW = _Lshl(OpSize::i64Bit, Top, _Constant(11));
|
||||
@@ -706,18 +720,21 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
// There's no load Status Word instruction but you can load it through frstor
|
||||
// or fldenv.
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResult(GPRClass, Op, StatusWord, -1);
|
||||
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
_SetRoundingMode(Zero, false, Zero);
|
||||
}
|
||||
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _Constant(OpSize::i16Bit, 0x037F);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Set top to zero
|
||||
SetX87Top(Zero);
|
||||
@@ -782,13 +799,14 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
auto AllOneConst = _Constant(0xffff'ffff'ffff'ffffull);
|
||||
|
||||
Ref SrcCond = SelectCC(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
|
||||
Ref VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
_F80VBSLStack(16, VecCond, Op->OP & 7, 0);
|
||||
Ref VecCond = _VDupFromGPR(OpSize::i128Bit, OpSize::i64Bit, SrcCond);
|
||||
_F80VBSLStack(OpSize::i128Bit, VecCond, Op->OP & 7, 0);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto a = _ReadStackValue(0);
|
||||
Ref Result = ReducedPrecisionMode ? _VExtractToGPR(8, 8, a, 0) : _VExtractToGPR(16, 8, a, 1);
|
||||
Ref Result =
|
||||
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
|
||||
@@ -810,4 +828,14 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) {
|
||||
auto Top = _ReadStackValue(0);
|
||||
|
||||
_PopStackDestroy();
|
||||
auto Exp = _F80XTRACT_EXP(Top);
|
||||
auto Sig = _F80XTRACT_SIG(Top);
|
||||
_PushStack(Exp, Exp, OpSize::f80Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::f80Bit, true);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -22,58 +23,28 @@ class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
// Functions in X87.cpp (no change required)
|
||||
// GetX87Top
|
||||
// SetX87ValidTag
|
||||
// GetX87ValidTag
|
||||
// GetX87Tag (will need changing once special tag handling is implemented)
|
||||
// SetX87FTW
|
||||
// GetX87FTW (will need changing once special tag handling is implemented)
|
||||
// SetX87Top
|
||||
// X87ModifySTP
|
||||
// EMMS
|
||||
// FFREE
|
||||
// FNSTENV
|
||||
// FSTCW
|
||||
// LDSW
|
||||
// FNSTSW
|
||||
// FXCH
|
||||
// FCMOV
|
||||
// FST(register to register)
|
||||
// FCHS
|
||||
|
||||
void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
// Init host rounding mode to zero
|
||||
auto Zero = _Constant(0);
|
||||
_SetRoundingMode(Zero, false, Zero);
|
||||
|
||||
// Call generic version
|
||||
FNINIT(Op);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size)), Size, MEM_OFFSET_SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
@@ -82,59 +53,59 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
// F64 ops
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, size_t Width) {
|
||||
size_t ReadWidth = (Width == 80) ? 16 : Width / 8;
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
if (Width == 32) {
|
||||
ConvertedData = _Float_FToF(8, 4, Data);
|
||||
} else if (Width == 80) {
|
||||
ConvertedData = _F80CVT(8, Data);
|
||||
if (Width == OpSize::i32Bit) {
|
||||
ConvertedData = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Data);
|
||||
} else if (Width == OpSize::f80Bit) {
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, Data);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(8, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, 8, true);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), 8);
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
|
||||
converted = _F80BCDStore(converted);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
|
||||
auto Data = _VCastFromGPR(8, 8, _Constant(Num));
|
||||
_PushStack(Data, Data, 8, true);
|
||||
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, _Constant(Num));
|
||||
_PushStack(Data, Data, OpSize::i64Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
size_t ReadWidth = GetSrcSize(Op);
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
if (ReadWidth == 2) {
|
||||
Data = _Sbfe(OpSize::i64Bit, ReadWidth * 8, 0, Data);
|
||||
if (ReadWidth == OpSize::i16Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
auto ConvertedData = _Float_FromGPR_S(8, ReadWidth == 4 ? 4 : 8, Data);
|
||||
auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, size_t Width) {
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i64Bit, true, Width / 8);
|
||||
_StoreStackMemory(Mem, OpSize::i64Bit, true, Width);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
@@ -142,22 +113,22 @@ void OpDispatchBuilder::FSTF64(OpcodeArgs, size_t Width) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref data = _ReadStackValue(0);
|
||||
if (Truncate) {
|
||||
data = _Float_ToGPR_ZS(Size == 4 ? 4 : 8, 8, data);
|
||||
data = _Float_ToGPR_ZS(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
} else {
|
||||
data = _Float_ToGPR_S(Size == 4 ? 4 : 8, 8, data);
|
||||
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FADDF64(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto Offset = Op->OP & 7;
|
||||
auto St0 = 0;
|
||||
@@ -177,15 +148,17 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -193,7 +166,7 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
}
|
||||
|
||||
// FIXME: following is very similar to FADDF64
|
||||
void OpDispatchBuilder::FMULF64(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto offset = Op->OP & 7;
|
||||
auto st0 = 0;
|
||||
@@ -213,15 +186,17 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -232,7 +207,7 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FDIVF64(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto offset = Op->OP & 7;
|
||||
const auto st0 = 0;
|
||||
@@ -260,19 +235,21 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, size_t Width, bool Integer, bool Rev
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
|
||||
}
|
||||
Arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, Arg);
|
||||
} else if (Width == 32) {
|
||||
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _Float_FToF(8, 4, Arg);
|
||||
} else if (Width == 64) {
|
||||
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -287,7 +264,7 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, size_t Width, bool Integer, bool Rev
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSUBF64(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
@@ -315,19 +292,21 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, size_t Width, bool Integer, bool Rev
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -348,11 +327,10 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
|
||||
// Now we do our comparison.
|
||||
_F80StackTest(0);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
@@ -360,22 +338,22 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispa
|
||||
// Implicit arg
|
||||
uint8_t offset = Op->OP & 7;
|
||||
b = _ReadStackValue(offset);
|
||||
} else {
|
||||
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
if (WhichFlags == FCOMIFlags::FLAGS_X87) {
|
||||
@@ -383,7 +361,6 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispa
|
||||
GetNZCV();
|
||||
|
||||
_F80CmpValue(b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
} else {
|
||||
HandleNZCVWrite();
|
||||
@@ -399,144 +376,37 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispa
|
||||
}
|
||||
}
|
||||
|
||||
// This function converts to F80 on save for compatibility
|
||||
void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
// 14 bytes for 16bit
|
||||
// 2 Bytes : FCW
|
||||
// 2 Bytes : FSW
|
||||
// 2 bytes : FTW
|
||||
// 2 bytes : Instruction offset
|
||||
// 2 bytes : Instruction CS selector
|
||||
// 2 bytes : Data offset
|
||||
// 2 bytes : Data selector
|
||||
void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
// Split node into SIG and EXP while handling the special zero case.
|
||||
// i.e. if val == 0.0, then sig = 0.0, exp = -inf
|
||||
// if val == -0.0, then sig = -0.0, exp = -inf
|
||||
// otherwise we just extract the 64-bit sig and exp as normal.
|
||||
Ref Node = _ReadStackValue(0);
|
||||
|
||||
// 28 bytes for 32bit
|
||||
// 4 bytes : FCW
|
||||
// 4 bytes : FSW
|
||||
// 4 bytes : FTW
|
||||
// 4 bytes : Instruction pointer
|
||||
// 2 bytes : instruction pointer selector
|
||||
// 2 bytes : Opcode
|
||||
// 4 bytes : data pointer offset
|
||||
// 4 bytes : data pointer selector
|
||||
Ref Gpr = _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Node, 0);
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
// zero case
|
||||
Ref ExpZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, _Constant(0xfff0'0000'0000'0000UL));
|
||||
Ref SigZV = Node;
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
// non zero case
|
||||
Ref ExpNZ = _Bfe(OpSize::i64Bit, 11, 52, Gpr);
|
||||
ExpNZ = _Sub(OpSize::i64Bit, ExpNZ, _Constant(1023));
|
||||
Ref ExpNZV = _Float_FromGPR_S(OpSize::i64Bit, OpSize::i64Bit, ExpNZ);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
Ref SigNZ = _And(OpSize::i64Bit, Gpr, _Constant(0x800f'ffff'ffff'ffffLL));
|
||||
SigNZ = _Or(OpSize::i64Bit, SigNZ, _Constant(0x3ff0'0000'0000'0000LL));
|
||||
Ref SigNZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, SigNZ);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
// Comparison and select to push onto stack
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Gpr, _Constant(0x7fff'ffff'ffff'ffffUL));
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, ExpZV, ExpNZV);
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTTo(data, 8);
|
||||
_StoreMem(FPRClass, 16, data, Mem, _Constant((Size * 7) + (i * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
Ref data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTTo(data, 8);
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, data, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, topBytes, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINITF64(Op);
|
||||
_PopStackDestroy();
|
||||
_PushStack(Exp, Exp, OpSize::i64Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::i64Bit, true);
|
||||
}
|
||||
|
||||
// This function converts from F80 on load for compatibility
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
const auto Size = GetSrcSize(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = NewFCW;
|
||||
auto roundShift = _Constant(10);
|
||||
auto roundMask = _Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, 16, Mem, _Constant((Size * 7) + (i * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(8, Reg);
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
|
||||
Ref Reg = _LoadMem(FPRClass, 8, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, 2, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
Reg = _F80CVT(8, Reg); // Convert to double precision
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -14,12 +14,14 @@ namespace FEXCore::X86Tables {
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializePrimaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode);
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
InitializeSecondaryGroupTables(Mode);
|
||||
InitializePrimaryGroupTables(Mode);
|
||||
InitializeH0F3ATables(Mode);
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/BaseTables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -144,7 +145,7 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
// These three are all X87 instructions
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9C, 1, X86InstInfo{"PUSHF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -236,6 +237,7 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), BaseOpTable, std::size(BaseOpTable));
|
||||
IR::InstallToTable(Table, IR::OpDispatch_BaseOpTable);
|
||||
|
||||
return Table;
|
||||
}();
|
||||
@@ -301,9 +303,11 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_64, std::size(BaseOpTable_64));
|
||||
IR::InstallToTable(BaseOps, IR::OpDispatch_BaseOpTable_64);
|
||||
}
|
||||
else {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_32, std::size(BaseOpTable_32));
|
||||
IR::InstallToTable(BaseOps, IR::OpDispatch_BaseOpTable_32);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/DDDTables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -54,6 +55,8 @@ std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_DDDTable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -1,37 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: frontend|x86-tables
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct EVEXTable[] = {
|
||||
{0x10, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x11, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"VBROADCASTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x19, 1, X86InstInfo{"VBROADCASTD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x1A, 1, X86InstInfo{"VBROADCASTSD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x1B, 1, X86InstInfo{"VBROADCASTF64X4", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x28, 1, X86InstInfo{"VMOVAPS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x29, 1, X86InstInfo{"VMOVAPS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"VBROADCASTQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"VMOVDQU64", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x73, 1, X86InstInfo{"VPSLLDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x7F, 1, X86InstInfo{"VMOVDQU64", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0xE7, 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), EVEXTable, std::size(EVEXTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/H0F38Tables.h"
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
@@ -119,6 +120,8 @@ std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), H0F38Table, std::size(H0F38Table));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F38Table);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/H0F3ATables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -20,47 +21,60 @@ constexpr uint16_t PF_3A_66 = 1;
|
||||
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
auto TableGen = []<uint16_t REX>() consteval {
|
||||
constexpr U16U8InfoStruct Table[] = {
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
constexpr auto H0F3ATable_IgnoresREX0 = TableGen.template operator()<0>();
|
||||
constexpr auto H0F3ATable_IgnoresREX1 = TableGen.template operator()<1>();
|
||||
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX0.at(0), H0F3ATable_IgnoresREX0.size());
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX1.at(0), H0F3ATable_IgnoresREX1.size());
|
||||
|
||||
constexpr U16U8InfoStruct TableNeedsREX[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
GenerateTable(&Table.at(0), TableNeedsREX, std::size(TableNeedsREX));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableIgnoreREX);
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableNeedsREX0);
|
||||
|
||||
GenerateTable(&Table.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
static constexpr U16U8InfoStruct H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
@@ -69,6 +83,7 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable_64, std::size(H0F3ATable_64));
|
||||
IR::InstallToTable(H0F3ATableOps, IR::OpDispatch_H0F3ATable_64);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/PrimaryGroupTables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -144,6 +145,8 @@ std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps = []() co
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_PrimaryGroupTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/SecondaryGroupTables.h"
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
@@ -488,7 +489,15 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryGroupTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeSecondaryGroupTables(Context::OperatingMode Mode) {
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
IR::InstallToTable(SecondInstGroupOps, IR::OpDispatch_SecondaryGroupTables_64);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/SecondaryModRMTables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -56,6 +57,8 @@ std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []()
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryModRMTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/SecondaryTables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -270,6 +271,8 @@ auto BaseOpsLambda = []() consteval {
|
||||
|
||||
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_TwoByteOpTable);
|
||||
|
||||
return Table;
|
||||
};
|
||||
|
||||
@@ -297,7 +300,7 @@ std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
|
||||
{0x2E, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x30, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -359,6 +362,7 @@ std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
|
||||
|
||||
GenerateTableWithCopy(&Table.at(0), RepModOpTable, std::size(RepModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryRepModTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -383,7 +387,7 @@ std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
|
||||
{0x2E, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x30, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -440,6 +444,7 @@ std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
|
||||
|
||||
GenerateTableWithCopy(&Table.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryRepNEModTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -595,6 +600,7 @@ std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps = []() consteval
|
||||
|
||||
GenerateTableWithCopy(&Table.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryOpSizeModTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -620,12 +626,16 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
|
||||
IR::InstallToTable(SecondBaseOps, IR::OpDispatch_TwoByteOpTable_64);
|
||||
}
|
||||
else {
|
||||
LateInitCopyTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
|
||||
IR::InstallToTable(SecondBaseOps, IR::OpDispatch_TwoByteOpTable_32);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/VEXTables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -489,6 +490,8 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), VEXTable, std::size(VEXTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_VEXTable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -521,6 +524,7 @@ std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps = []() conste
|
||||
|
||||
GenerateTable(&Table.at(0), VEXGroupTable, std::size(VEXGroupTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_VEXGroupTable);
|
||||
return Table;
|
||||
}();
|
||||
}
|
||||
@@ -225,7 +225,6 @@ enum InstType {
|
||||
TYPE_SECONDARY_TABLE_PREFIX,
|
||||
TYPE_X87_TABLE_PREFIX,
|
||||
TYPE_VEX_TABLE_PREFIX,
|
||||
TYPE_XOP_TABLE_PREFIX,
|
||||
TYPE_INST,
|
||||
TYPE_X87 = TYPE_INST,
|
||||
TYPE_INVALID,
|
||||
@@ -466,16 +465,6 @@ constexpr size_t MAX_VEX_TABLE_SIZE = (1 << 13);
|
||||
// group select (3 bits for now) | ModRM opcode (3 bits)
|
||||
constexpr size_t MAX_VEX_GROUP_TABLE_SIZE = (1 << 7);
|
||||
|
||||
// XOP
|
||||
// group (2 bits for now) | vex.pp (2 bits) | opcode (8bit)
|
||||
constexpr size_t MAX_XOP_TABLE_SIZE = (1 << 13);
|
||||
|
||||
// XOP group ops
|
||||
// group select (2 bits for now) | modrm opcode (3 bits)
|
||||
constexpr size_t MAX_XOP_GROUP_TABLE_SIZE = (1 << 6);
|
||||
|
||||
constexpr size_t MAX_EVEX_TABLE_SIZE = 256;
|
||||
|
||||
extern std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps;
|
||||
extern std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps;
|
||||
extern std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps;
|
||||
@@ -494,13 +483,6 @@ extern std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps;
|
||||
extern std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps;
|
||||
extern std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps;
|
||||
|
||||
// XOP
|
||||
extern std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps;
|
||||
extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
|
||||
// EVEX
|
||||
extern std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
|
||||
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
@@ -518,7 +500,10 @@ constexpr static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInf
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_MSG_A_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if (FinalTable[OpNum + i].OpcodeDispatcher) {
|
||||
LOGMAN_MSG_A_FMT("Already installed an OpcodeDispatcher for 0x{:x}", OpNum + i);
|
||||
}
|
||||
FinalTable[OpNum + i] = Info;
|
||||
}
|
||||
@@ -533,7 +518,7 @@ constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86T
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_MSG_A_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if (Info.Type == TYPE_COPY_OTHER) {
|
||||
FinalTable[OpNum + i] = OtherLocal[OpNum + i];
|
||||
@@ -568,7 +553,7 @@ constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86Tables
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_MSG_A_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if ((OpNum & 0b11'000'000) == 0b11'000'000) {
|
||||
// If the mod field is 0b11 then it is a regular op
|
||||
|
||||
@@ -1,143 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: frontend|x86-tables
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> Table{};
|
||||
#define OPD(group, pp, opcode) ( (group << 10) | (pp << 8) | (opcode))
|
||||
constexpr uint16_t XOP_GROUP_8 = 0;
|
||||
constexpr uint16_t XOP_GROUP_9 = 1;
|
||||
constexpr uint16_t XOP_GROUP_A = 2;
|
||||
|
||||
constexpr U16U8InfoStruct XOPTable[] = {
|
||||
// Group 8
|
||||
{OPD(XOP_GROUP_8, 0, 0x85), 1, X86InstInfo{"VPMAXSSWW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x86), 1, X86InstInfo{"VPMACSSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x87), 1, X86InstInfo{"VPMAXSSDQL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0x8E), 1, X86InstInfo{"VPMACSSDD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x8F), 1, X86InstInfo{"VPMACSSDQH", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0x95), 1, X86InstInfo{"VPMAXSWW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x96), 1, X86InstInfo{"VPMAXSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x97), 1, X86InstInfo{"VPMAXSDQL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0x9E), 1, X86InstInfo{"VPMACSDD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x9F), 1, X86InstInfo{"VPMACSDQH", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xA2), 1, X86InstInfo{"VPCMOV", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xA3), 1, X86InstInfo{"VPPERM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xA6), 1, X86InstInfo{"VPMADCSSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xB6), 1, X86InstInfo{"VPMADCSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xC0), 1, X86InstInfo{"VPROTB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xC1), 1, X86InstInfo{"VPROTW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xC2), 1, X86InstInfo{"VPROTD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xC3), 1, X86InstInfo{"VPROTQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xCC), 1, X86InstInfo{"VPCOMccB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xCD), 1, X86InstInfo{"VPCOMccW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xCE), 1, X86InstInfo{"VPCOMccD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xCF), 1, X86InstInfo{"VPCOMccQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xEC), 1, X86InstInfo{"VPCOMccUB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xED), 1, X86InstInfo{"VPCOMccUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xEE), 1, X86InstInfo{"VPCOMccUD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xEF), 1, X86InstInfo{"VPCOMccUQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 9
|
||||
{OPD(XOP_GROUP_9, 0, 0x01), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 1
|
||||
{OPD(XOP_GROUP_9, 0, 0x02), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 2
|
||||
{OPD(XOP_GROUP_9, 0, 0x12), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 3
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0x80), 1, X86InstInfo{"VFRZPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x81), 1, X86InstInfo{"VFRCZPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x82), 1, X86InstInfo{"VFRCZSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x83), 1, X86InstInfo{"VFRCZSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0x90), 1, X86InstInfo{"VPROTB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x91), 1, X86InstInfo{"VPROTW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x92), 1, X86InstInfo{"VPROTD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x93), 1, X86InstInfo{"VRPTOQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x94), 1, X86InstInfo{"VPSHLB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x95), 1, X86InstInfo{"VPSHLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x96), 1, X86InstInfo{"VPSHLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x97), 1, X86InstInfo{"VPSHLQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0x98), 1, X86InstInfo{"VPSHAB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x99), 1, X86InstInfo{"VPSHAW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x9A), 1, X86InstInfo{"VPSHAD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x9B), 1, X86InstInfo{"VPSHAQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0xC1), 1, X86InstInfo{"VPHADDBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC2), 1, X86InstInfo{"VPHADDBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC3), 1, X86InstInfo{"VPHADDBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC6), 1, X86InstInfo{"VPHADDWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC7), 1, X86InstInfo{"VPHADDWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xCB), 1, X86InstInfo{"VPHADDDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0xD1), 1, X86InstInfo{"VPHADDUBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD2), 1, X86InstInfo{"VPHADDUBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD3), 1, X86InstInfo{"VPHADDUBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD6), 1, X86InstInfo{"VPHADDUWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD7), 1, X86InstInfo{"VPHADDUWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xDB), 1, X86InstInfo{"VPHADDUDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0xE1), 1, X86InstInfo{"VPHSUBBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xE2), 1, X86InstInfo{"VPHSUBBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xE3), 1, X86InstInfo{"VPHSUBDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group A
|
||||
{OPD(XOP_GROUP_A, 0, 0x10), 1, X86InstInfo{"BEXTR", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_A, 0, 0x12), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 4
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), XOPTable, std::size(XOPTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> Table{};
|
||||
#define OPD(subgroup, opcode) (((subgroup - 1) << 3) | (opcode))
|
||||
constexpr U8U8InfoStruct XOPGroupTable[] = {
|
||||
// Group 1
|
||||
{OPD(1, 1), 1, X86InstInfo{"BLCFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 2), 1, X86InstInfo{"BLSFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 3), 1, X86InstInfo{"BLCS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 4), 1, X86InstInfo{"TZMSK", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 5), 1, X86InstInfo{"BLCIC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 6), 1, X86InstInfo{"BLSIC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 7), 1, X86InstInfo{"T1MSKC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 2
|
||||
{OPD(2, 1), 1, X86InstInfo{"BLCMSK", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 6), 1, X86InstInfo{"BLCI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 3
|
||||
{OPD(3, 0), 1, X86InstInfo{"LLWPCB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 1), 1, X86InstInfo{"SLWPCB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 4
|
||||
{OPD(4, 0), 1, X86InstInfo{"LWPINS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(4, 1), 1, X86InstInfo{"LWPVAL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), XOPGroupTable, std::size(XOPGroupTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -17,9 +17,34 @@
|
||||
#include <shared_mutex>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugData;
|
||||
}
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
};
|
||||
|
||||
struct DebugDataGuestOpcode {
|
||||
uint64_t GuestEntryOffset;
|
||||
ptrdiff_t HostEntryOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Contains debug data for a block of code for later debugger analysis
|
||||
*
|
||||
* Needs to remain around for as long as the code could be executed at least
|
||||
*/
|
||||
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
fextl::vector<DebugDataSubblock> Subblocks;
|
||||
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
|
||||
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
|
||||
};
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
@@ -548,13 +548,16 @@ protected:
|
||||
|
||||
// This must directly match bytes to the named opsize.
|
||||
// Implicit sized IR operations does math to get between sizes.
|
||||
enum OpSize : uint8_t {
|
||||
enum class OpSize : uint8_t {
|
||||
iUnsized = 0,
|
||||
i8Bit = 1,
|
||||
i16Bit = 2,
|
||||
i32Bit = 4,
|
||||
i64Bit = 8,
|
||||
f80Bit = 10,
|
||||
i128Bit = 16,
|
||||
i256Bit = 32,
|
||||
iInvalid = 0xFF,
|
||||
};
|
||||
|
||||
enum class FloatCompareOp : uint8_t {
|
||||
@@ -578,16 +581,71 @@ enum class ShiftType : uint8_t {
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
switch (Size) {
|
||||
case 0: return OpSize::iUnsized;
|
||||
case 1: return OpSize::i8Bit;
|
||||
case 2: return OpSize::i16Bit;
|
||||
case 4: return OpSize::i32Bit;
|
||||
case 8: return OpSize::i64Bit;
|
||||
case 10: return OpSize::f80Bit;
|
||||
case 16: return OpSize::i128Bit;
|
||||
case 32: return OpSize::i256Bit;
|
||||
case 0xFF: return OpSize::iInvalid;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline uint8_t OpSizeToSize(IR::OpSize Size) {
|
||||
switch (Size) {
|
||||
case OpSize::iUnsized: return 0;
|
||||
case OpSize::i8Bit: return 1;
|
||||
case OpSize::i16Bit: return 2;
|
||||
case OpSize::i32Bit: return 4;
|
||||
case OpSize::i64Bit: return 8;
|
||||
case OpSize::f80Bit: return 10;
|
||||
case OpSize::i128Bit: return 16;
|
||||
case OpSize::i256Bit: return 32;
|
||||
case OpSize::iInvalid: return 0xFF;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
static inline uint16_t OpSizeAsBits(IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::OpSizeToSize(Size) * 8u;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_integral_v<T>)
|
||||
static inline OpSize operator<<(IR::OpSize Size, T Shift) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) << Shift);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_integral_v<T>)
|
||||
static inline OpSize operator>>(IR::OpSize Size, T Shift) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) >> Shift);
|
||||
}
|
||||
|
||||
static inline OpSize operator/(IR::OpSize Size, IR::OpSize Divisor) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) / IR::OpSizeToSize(Divisor));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_integral_v<T>)
|
||||
static inline OpSize operator/(IR::OpSize Size, T Divisor) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) / Divisor);
|
||||
}
|
||||
|
||||
static inline uint8_t NumElements(IR::OpSize RegisterSize, IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::OpSizeToSize(RegisterSize) / IR::OpSizeToSize(ElementSize);
|
||||
}
|
||||
|
||||
#define IROP_ENUM
|
||||
#define IROP_STRUCTS
|
||||
#define IROP_SIZES
|
||||
|
||||
+481
-427
File diff suppressed because it is too large.
Load diff
@@ -77,6 +77,8 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else if (Arg == PREDClass.Val) {
|
||||
*out << "PRED";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
@@ -98,6 +100,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::PREDClass.Val: *out << "(PRED"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
@@ -112,17 +115,17 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
}
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
uint32_t NumElements = IROp->Size;
|
||||
if (!IROp->ElementSize) {
|
||||
auto ElementSize = IROp->ElementSize;
|
||||
uint32_t NumElements = 0;
|
||||
if (IROp->ElementSize == OpSize::iUnsized) {
|
||||
ElementSize = IROp->Size;
|
||||
}
|
||||
|
||||
if (ElementSize) {
|
||||
NumElements /= ElementSize;
|
||||
if (ElementSize != OpSize::iUnsized) {
|
||||
NumElements = IR::NumElements(IROp->Size, ElementSize);
|
||||
}
|
||||
|
||||
*out << " i" << std::dec << (ElementSize * 8);
|
||||
*out << " i" << std::dec << IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
if (NumElements > 1) {
|
||||
*out << "v" << std::dec << NumElements;
|
||||
@@ -206,6 +209,22 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
return "x87_log10_2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG_2:
|
||||
return "x87_log2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32:
|
||||
return "cvtmax_f32_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32_UPPER:
|
||||
return "cvtmax_f32_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I64:
|
||||
return "cvtmax_f32_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32:
|
||||
return "cvtmax_f64_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32_UPPER:
|
||||
return "cvtmax_f64_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I64:
|
||||
return "cvtmax_f64_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I32:
|
||||
return "cvtmax_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
|
||||
return "cvtmax_i64";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
}
|
||||
@@ -294,14 +313,14 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
AddIndent();
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
uint32_t NumElements = IROp->Size;
|
||||
if (!IROp->ElementSize) {
|
||||
auto ElementSize = IROp->ElementSize;
|
||||
uint8_t NumElements = 0;
|
||||
if (IROp->ElementSize != OpSize::iUnsized) {
|
||||
ElementSize = IROp->Size;
|
||||
}
|
||||
|
||||
if (ElementSize) {
|
||||
NumElements /= ElementSize;
|
||||
if (ElementSize != OpSize::iUnsized) {
|
||||
NumElements = IR::NumElements(IROp->Size, ElementSize);
|
||||
}
|
||||
|
||||
*out << "%" << std::dec << ID;
|
||||
@@ -324,7 +343,7 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
}
|
||||
}
|
||||
|
||||
*out << " i" << std::dec << (ElementSize * 8);
|
||||
*out << " i" << std::dec << IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
if (NumElements > 1) {
|
||||
*out << "v" << std::dec << NumElements;
|
||||
@@ -333,17 +352,17 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
*out << " = ";
|
||||
} else {
|
||||
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
if (!IROp->ElementSize) {
|
||||
auto ElementSize = IROp->ElementSize;
|
||||
if (IROp->ElementSize == OpSize::iUnsized) {
|
||||
ElementSize = IROp->Size;
|
||||
}
|
||||
uint32_t NumElements = 0;
|
||||
if (ElementSize) {
|
||||
NumElements = IROp->Size / ElementSize;
|
||||
if (ElementSize != OpSize::iUnsized) {
|
||||
NumElements = IR::NumElements(IROp->Size, ElementSize);
|
||||
}
|
||||
|
||||
*out << "(%" << std::dec << ID << ' ';
|
||||
*out << 'i' << std::dec << (ElementSize * 8);
|
||||
*out << 'i' << std::dec << IR::OpSizeAsBits(ElementSize);
|
||||
if (NumElements > 1) {
|
||||
*out << 'v' << std::dec << NumElements;
|
||||
}
|
||||
|
||||
@@ -41,6 +41,7 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case PREDClass:
|
||||
case InvalidClass: return Class;
|
||||
default: break;
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
|
||||
@@ -9,9 +10,9 @@
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <new>
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
|
||||
@@ -45,6 +46,37 @@ public:
|
||||
}
|
||||
void ResetWorkingList();
|
||||
|
||||
// Predicate Cache Implementation
|
||||
// This lives here rather than OpcodeDispatcher because x87StackOptimization Pass
|
||||
// also needs it.
|
||||
struct PredicateKey {
|
||||
ARMEmitter::PredicatePattern Pattern;
|
||||
OpSize Size;
|
||||
bool operator==(const PredicateKey& rhs) const = default;
|
||||
};
|
||||
|
||||
struct PredicateKeyHash {
|
||||
size_t operator()(const PredicateKey& key) const {
|
||||
return FEXCore::ToUnderlying(key.Pattern) + (FEXCore::ToUnderlying(key.Size) * FEXCore::ToUnderlying(OpSize::iInvalid));
|
||||
}
|
||||
};
|
||||
fextl::unordered_map<PredicateKey, Ref, PredicateKeyHash> InitPredicateCache;
|
||||
|
||||
Ref InitPredicateCached(OpSize Size, ARMEmitter::PredicatePattern Pattern) {
|
||||
PredicateKey Key {Pattern, Size};
|
||||
auto ValIt = InitPredicateCache.find(Key);
|
||||
if (ValIt == InitPredicateCache.end()) {
|
||||
auto Predicate = _InitPredicate(Size, static_cast<uint8_t>(FEXCore::ToUnderlying(Pattern)));
|
||||
InitPredicateCache[Key] = Predicate;
|
||||
return Predicate;
|
||||
}
|
||||
return ValIt->second;
|
||||
}
|
||||
|
||||
void ResetInitPredicateCache() {
|
||||
InitPredicateCache.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* @name IR allocation routines
|
||||
*
|
||||
@@ -59,12 +91,12 @@ public:
|
||||
#define IROP_ALLOCATE_HELPERS
|
||||
#define IROP_DISPATCH_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
IRPair<IROp_Constant> _Constant(uint8_t Size, uint64_t Constant) {
|
||||
IRPair<IROp_Constant> _Constant(IR::OpSize Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
uint64_t Mask = ~0ULL >> (64 - Size);
|
||||
uint64_t Mask = ~0ULL >> (64 - IR::OpSizeAsBits(Size));
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size / 8;
|
||||
Op.first->Header.ElementSize = Size / 8;
|
||||
Op.first->Header.Size = Size;
|
||||
Op.first->Header.ElementSize = Size;
|
||||
return Op;
|
||||
}
|
||||
IRPair<IROp_Jump> _Jump() {
|
||||
@@ -77,24 +109,24 @@ public:
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
// TODO: Work to remove this implicit sized Select implementation.
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, uint8_t CompareSize = 0) {
|
||||
if (CompareSize == 0) {
|
||||
CompareSize = std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, IR::OpSize CompareSize = OpSize::iUnsized) {
|
||||
if (CompareSize == OpSize::iUnsized) {
|
||||
CompareSize = std::max(OpSize::i32Bit, std::max(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
}
|
||||
|
||||
return _Select(IR::SizeToOpSize(std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa2), GetOpSize(ssa3)))),
|
||||
IR::SizeToOpSize(CompareSize), CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
return _Select(std::max(OpSize::i32Bit, std::max(GetOpSize(ssa2), GetOpSize(ssa3))), CompareSize, CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMemTSO>
|
||||
_StoreMemTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
Ref Invalid() {
|
||||
@@ -343,8 +375,13 @@ protected:
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
// MMX State can be either MMX (for 64bit) or x87 FPU (for 80bit)
|
||||
enum { MMXState_MMX, MMXState_X87 } MMXState = MMXState_MMX;
|
||||
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
virtual void RecordX87Use() {}
|
||||
virtual void ChgStateX87_MMX() {}
|
||||
virtual void ChgStateMMX_X87() {}
|
||||
virtual void SaveNZCV(IROps Op) {}
|
||||
|
||||
Ref CurrentWriteCursor = nullptr;
|
||||
|
||||
@@ -70,8 +70,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass());
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
|
||||
namespace FEXCore {
|
||||
class CPUIDEmu;
|
||||
}
|
||||
struct HostFeatures;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class IntrusivePooledAllocator;
|
||||
@@ -18,9 +19,8 @@ class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
|
||||
@@ -18,18 +18,13 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <string.h>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
uint64_t NumBits = Op->Size * 8;
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
@@ -52,17 +47,6 @@ static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
return ARMEmitter::Emitter::IsImmLogical(imm, width);
|
||||
}
|
||||
|
||||
static bool IsBfeAlreadyDone(IREmitter* IREmit, OrderedNodeWrapper src, uint64_t Width) {
|
||||
auto IROp = IREmit->GetOpHeader(src);
|
||||
if (IROp->Op == OP_BFE) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
if (Width >= Op->Width) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID)
|
||||
@@ -74,13 +58,40 @@ public:
|
||||
private:
|
||||
void HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR);
|
||||
void ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp);
|
||||
void ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR);
|
||||
|
||||
fextl::unordered_map<uint64_t, Ref> ConstPool;
|
||||
|
||||
bool SupportsTSOImm9 {};
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
|
||||
template<class F>
|
||||
bool InlineIf(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index, F Filter) {
|
||||
uint64_t Constant;
|
||||
if (!IREmit->IsValueConstant(IROp->Args[Index], &Constant) || !Filter(Constant)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[Index]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Index, IREmit->_InlineConstant(Constant));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Inline(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, [](uint64_t _) { return true; });
|
||||
}
|
||||
|
||||
bool InlineIfZero(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, [](uint64_t X) { return X == 0; });
|
||||
}
|
||||
|
||||
bool InlineIfLargeAddSub(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
auto Filter = [&IROp](uint64_t X) {
|
||||
return ARMEmitter::IsImmAddSub(X) && IROp->Size >= OpSize::i32Bit;
|
||||
};
|
||||
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
}
|
||||
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IROp_Header* IROp, OrderedNodeWrapper Offset,
|
||||
MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
uint64_t Imm {};
|
||||
@@ -96,7 +107,7 @@ private:
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
bool IsExtended = (Imm & (IROp->Size - 1)) == 0 && Imm / IROp->Size <= 4095;
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(IROp->Size) - 1)) == 0 && Imm / IR::OpSizeToSize(IROp->Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
if (IsSIMM9 || IsExtended) {
|
||||
@@ -109,24 +120,98 @@ private:
|
||||
|
||||
// Constants are pooled per block.
|
||||
void ConstProp::HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
const uint32_t SSACount = CurrentIR.GetSSACount();
|
||||
|
||||
// Allocation/initialization deferred until first use, since many multiblocks
|
||||
// don't have constants leftover after all inlining.
|
||||
fextl::vector<Ref> Remap {};
|
||||
|
||||
struct Entry {
|
||||
int64_t Value;
|
||||
Ref R;
|
||||
};
|
||||
|
||||
|
||||
fextl::vector<Entry> Pool {};
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
Pool.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
auto it = ConstPool.find(Op->Constant);
|
||||
bool Found = false;
|
||||
|
||||
if (it != ConstPool.end()) {
|
||||
auto CodeIter = CurrentIR.at(CodeNode);
|
||||
IREmit->ReplaceUsesWithAfter(CodeNode, it->second, CodeIter);
|
||||
} else {
|
||||
ConstPool[Op->Constant] = CodeNode;
|
||||
// Search for the constant. This is O(n^2) but n is small since it's
|
||||
// local and most constants are inlined. In practice, it ends up much
|
||||
// faster than a hash table.
|
||||
for (auto K : Pool) {
|
||||
if (K.Value == Op->Constant) {
|
||||
uint32_t Value = CurrentIR.GetID(CodeNode).Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "def not yet remapped");
|
||||
|
||||
if (Remap.empty()) {
|
||||
Remap.resize(SSACount, nullptr);
|
||||
}
|
||||
|
||||
Remap[Value] = K.R;
|
||||
Found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!Found) {
|
||||
Pool.push_back({.Value = Op->Constant, .R = CodeNode});
|
||||
}
|
||||
} else if (!Remap.empty()) {
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t Value = IROp->Args[i].ID().Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "src not yet remapped");
|
||||
|
||||
Ref New = Remap[Value];
|
||||
if (New) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, New);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
ConstPool.clear();
|
||||
}
|
||||
}
|
||||
|
||||
// Helper to replace the destination of an instruction with one of its sources,
|
||||
// to implement algebraic identities. This is surprisingly tricky due to
|
||||
// implicit masking in our IR.
|
||||
//
|
||||
// FEX's IR uses sized opcodes, matching arm64 semantics. 64-bit opcodes do not
|
||||
// mask, whereas smaller opcodes mask/zero-extend from 32-bits. Therefore, if
|
||||
// the instruction is 32-bit, we need to mask the source for a sound
|
||||
// replacement, in case there was garbage in the upper bits.
|
||||
//
|
||||
// However, if that source is in turn written by a 32-bit instruction, it is
|
||||
// guaranteed to have already been masked, so we know there's no garbage and we
|
||||
// can avoid the zero-extension. This is the case 99% of the time, but the
|
||||
// masking here is correctness-bearing nevertheless (and new versions of Denuvo
|
||||
// break if you get this wrong!)
|
||||
static inline void ReplaceWithSource(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Idx) {
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[Idx]);
|
||||
|
||||
if (IROp->Size < OpSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == OpSize::i32Bit, "other sizes not here");
|
||||
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[Idx]);
|
||||
if (Header->Size > OpSize::i32Bit) {
|
||||
Arg = IREmit->_Bfe(OpSize::i32Bit, 32, 0, Arg);
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
@@ -143,7 +228,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
* here so we get the optimization for 32-bit adds too.
|
||||
*/
|
||||
if (Op->Header.Size == 4) {
|
||||
if (Op->Header.Size == OpSize::i32Bit) {
|
||||
Constant1 = (int64_t)(int32_t)Constant1;
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
@@ -151,10 +236,14 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
break;
|
||||
} else if (IsConstant1 && IsConstant2 && IROp->Op == OP_SUB) {
|
||||
uint64_t NewConstant = (Constant1 - Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// So, negate the operation to negate (and inline) the constant.
|
||||
if (IROp->Op == OP_ADD) {
|
||||
@@ -175,6 +264,23 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
// Replace the second source with the negated constant.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant);
|
||||
}
|
||||
|
||||
if (!InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1) && (IROp->Op == OP_SUB || IROp->Op == OP_SUBWITHFLAGS)) {
|
||||
// TODO: Generalize this
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ADDNZCV: {
|
||||
InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_SUBNZCV: {
|
||||
if (!InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
// TODO: Generalize this
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
@@ -194,51 +300,38 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
bool Replaced = false;
|
||||
|
||||
// Order matter for short circuit evaluation, subsequent ifs read constant2.
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
uint64_t NewConstant = (Constant1 & Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (Constant2 == 1) {
|
||||
// happens from flag calcs
|
||||
auto val = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
|
||||
uint64_t Constant3;
|
||||
if (val->Op == OP_SELECT && IREmit->IsValueConstant(val->Args[2], &Constant2) && IREmit->IsValueConstant(val->Args[3], &Constant3) &&
|
||||
Constant2 == 1 && Constant3 == 0) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
Replaced = true;
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
Replaced = true;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | Constant2;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
// OR with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_XOR: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 ^ Constant2;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
// XOR with same value results to zero
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, IREmit->_Constant(0));
|
||||
} else {
|
||||
// XOR with zero results in the nonzero source
|
||||
bool Replaced = false;
|
||||
for (unsigned i = 0; i < 2; ++i) {
|
||||
if (!IREmit->IsValueConstant(IROp->Args[i], &Constant1)) {
|
||||
continue;
|
||||
@@ -249,13 +342,23 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[1 - i]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 1 - i);
|
||||
Replaced = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN:
|
||||
case OP_TESTNZ: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_NEG: {
|
||||
uint64_t Constant {};
|
||||
|
||||
@@ -265,39 +368,36 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ASHR:
|
||||
case OP_ROR: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHL: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t ShiftMask = IROp->Size == OpSize::i64Bit ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_LSHR: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
// The source is masked, which will produce a correctly masked
|
||||
// destination. Masking the destination without the source instead will
|
||||
// right-shift garbage into the upper bits instead of zeroes.
|
||||
Constant1 &= getMask(IROp);
|
||||
Constant2 &= (IROp->Size == 8 ? 63 : 31);
|
||||
uint64_t NewConstant = (Constant1 >> Constant2);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -305,46 +405,12 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
|
||||
// Is this value already BFE'd?
|
||||
if (IsBfeAlreadyDone(IREmit, Op->Src, Op->Width)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Src));
|
||||
break;
|
||||
}
|
||||
|
||||
// Is this value already ZEXT'd?
|
||||
if (Op->lsb == 0) {
|
||||
// LoadMem, LoadMemTSO & LoadContext ZExt
|
||||
auto source = Op->Src;
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (Op->Width >= (sourceHeader->Size * 8) &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)) {
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (IROp->Size <= 8 && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
if (IROp->Size <= OpSize::i64Bit && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
uint64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IROp->Size == CurrentIR.GetOp<IROp_Header>(IROp->Args[0])->Size && Op->Width == (IROp->Size * 8) && Op->lsb == 0) {
|
||||
// A BFE that extracts all bits results in original value
|
||||
// XXX - This is broken for now - see https://github.com/FEX-Emu/FEX/issues/351
|
||||
// IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
} else if (Op->Width == 1 && Op->lsb == 0) {
|
||||
// common from flag codegen
|
||||
auto val = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (val->Op == OP_SELECT && IREmit->IsValueConstant(val->Args[2], &Constant2) && IREmit->IsValueConstant(val->Args[3], &Constant3) &&
|
||||
Constant2 == 1 && Constant3 == 0) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
@@ -355,7 +421,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IROp->Size * 8;
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
uint64_t DestMask = DestSizeInBits == 64 ? ~0ULL : ((1ULL << DestSizeInBits) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
@@ -369,52 +435,26 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantDest {};
|
||||
uint64_t ConstantSrc {};
|
||||
bool DestIsConstant = IREmit->IsValueConstant(IROp->Args[0], &ConstantDest);
|
||||
bool SrcIsConstant = IREmit->IsValueConstant(IROp->Args[1], &ConstantSrc);
|
||||
|
||||
if (DestIsConstant && SrcIsConstant) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t NewConstant = ConstantDest & ~(SourceMask << Op->lsb);
|
||||
NewConstant |= (ConstantSrc & SourceMask) << Op->lsb;
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) {
|
||||
if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) {
|
||||
// We are trying to insert constant, if it is a bitfield of only set bits then we can orr or and it.
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t NewConstant = SourceMask << Op->lsb;
|
||||
|
||||
if (ConstantSrc & 1) {
|
||||
auto orr = IREmit->_Or(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
auto orr = IREmit->_Or(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, orr);
|
||||
} else {
|
||||
// We are wanting to clear the bitfield.
|
||||
auto andn = IREmit->_Andn(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
auto andn = IREmit->_Andn(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, andn);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MUL: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 * Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && std::popcount(Constant2) == 1) {
|
||||
if (IROp->Size == 4 || IROp->Size == 8) {
|
||||
uint64_t amt = std::countr_zero(Constant2);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto shift = IREmit->_Lshl(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(amt));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, shift);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
@@ -561,244 +601,101 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
break;
|
||||
}
|
||||
|
||||
default: break;
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT:
|
||||
case OP_RMIFNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
|
||||
void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
switch (IROp->Op) {
|
||||
case OP_LSHR:
|
||||
case OP_ASHR:
|
||||
case OP_ROR:
|
||||
case OP_LSHL: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
uint64_t AllOnes = IROp->Size == OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
// this shouldn't be here, but rather on the emitter themselves or the constprop transformation?
|
||||
if (IROp->Size <= 4) {
|
||||
Constant2 &= 31;
|
||||
} else {
|
||||
Constant2 &= 63;
|
||||
}
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[2], &Constant2) && IREmit->IsValueConstant(IROp->Args[3], &Constant3) &&
|
||||
(Constant2 == 1 || Constant2 == AllOnes) && Constant3 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
break;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
case OP_ADDNZCV:
|
||||
case OP_SUBNZCV:
|
||||
case OP_ADDWITHFLAGS:
|
||||
case OP_SUBWITHFLAGS: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
if (ARMEmitter::IsImmAddSub(Constant2) && IROp->Size >= 4) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
} else if (IROp->Op == OP_SUBNZCV || IROp->Op == OP_SUBWITHFLAGS || IROp->Op == OP_SUB) {
|
||||
// TODO: Generalize this
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT: {
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
if (InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
uint64_t AllOnes = IROp->Size == OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 0, [&AllOnes](uint64_t X) { return X == 1 || X == AllOnes; });
|
||||
}
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
break;
|
||||
}
|
||||
case OP_EXITFUNCTION: {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_RMIFNZCV: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1)) {
|
||||
if (IsImmLogical(Constant1, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[2], &Constant2) && IREmit->IsValueConstant(IROp->Args[3], &Constant3) &&
|
||||
(Constant2 == 1 || Constant2 == AllOnes) && Constant3 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT: {
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
uint64_t Constant0 {};
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1) && Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant0) && (Constant0 == 1 || Constant0 == AllOnes)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_EXITFUNCTION: {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) {
|
||||
if (!Inline(IREmit, CurrentIR, CodeNode, IROp, Op->NewRIP_Index)) {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant));
|
||||
} else {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(IR::SizeToOpSize(EO->Header.Size), EO->Offset));
|
||||
}
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(EO->Header.Size, EO->Offset));
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_AND:
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (IsImmLogical(Constant2, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
break;
|
||||
}
|
||||
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, IREmit->_InlineConstant(Constant));
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, IREmit->_InlineConstant(Constant));
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -807,13 +704,11 @@ void ConstProp::Run(IREmitter* IREmit) {
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
HandleConstantPools(IREmit, CurrentIR);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
}
|
||||
|
||||
ConstantInlining(IREmit, CurrentIR);
|
||||
HandleConstantPools(IREmit, IREmit->ViewIR());
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID) {
|
||||
|
||||
@@ -1,154 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Cross block store-after-store elimination
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
constexpr int PropagationRounds = 5;
|
||||
|
||||
// Return a bit representing a single GPR or FPR.
|
||||
static inline uint64_t RegBit(RegisterClassType Class, uint32_t Reg) {
|
||||
uint32_t AdjustedReg = (Class == FPRClass) ? (32 + Reg) : Reg;
|
||||
|
||||
return 1ULL << AdjustedReg;
|
||||
}
|
||||
|
||||
class DeadStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
void Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
struct ReadWriteKill {
|
||||
uint64_t reads {0};
|
||||
uint64_t writes {0};
|
||||
uint64_t kill {0};
|
||||
};
|
||||
|
||||
struct Info {
|
||||
ReadWriteKill reg;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead reg stores
|
||||
*
|
||||
* First pass computes which regs are read and written per block
|
||||
*
|
||||
* Second pass computes which regs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead regs across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
void DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::vector<Info> InfoMap(CurrentIR.GetSSACount());
|
||||
|
||||
// Pass 1
|
||||
// Compute regs read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
BlockInfo.reg.writes |= RegBit(Op->Class, Op->Reg);
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
BlockInfo.reg.reads |= RegBit(Op->Class, Op->Reg);
|
||||
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
|
||||
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
|
||||
|
||||
if (Op->Flags & (1u << X86State::RFLAG_PF_RAW_LOC)) {
|
||||
BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::PF_AS_GREG);
|
||||
}
|
||||
|
||||
if (Op->Flags & (1u << X86State::RFLAG_AF_RAW_LOC)) {
|
||||
BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::AF_AS_GREG);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute flags/registers that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < PropagationRounds; i++) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto CodeBlock = BlockIROp->C<IROp_CodeBlock>();
|
||||
|
||||
auto IROp = CurrentIR.GetNode(CurrentIR.GetNode(CodeBlock->Last)->Header.Previous)->Op(CurrentIR.GetData());
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
auto& TargetInfo = InfoMap[Op->Header.Args[0].ID().Value];
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.reg.kill = TargetInfo.reg.writes & ~(TargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
// If written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
|
||||
} else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
auto& TrueTargetInfo = InfoMap[Op->TrueBlock.ID().Value];
|
||||
auto& FalseTargetInfo = InfoMap[Op->FalseBlock.ID().Value];
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.reg.kill = TrueTargetInfo.reg.writes & ~(TrueTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
BlockInfo.reg.kill &= FalseTargetInfo.reg.writes & ~(FalseTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
// if written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.reg.kill & RegBit(Op->Class, Op->Reg)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination() {
|
||||
return fextl::make_unique<DeadStoreElimination>();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -79,12 +79,12 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto ID = CurrentIR.GetID(CodeNode);
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
HadError |= OpSize == 0;
|
||||
HadError |= OpSize == IR::OpSize::iInvalid;
|
||||
// Does the op have a destination of size 0?
|
||||
if (OpSize == 0) {
|
||||
if (OpSize == IR::OpSize::iInvalid) {
|
||||
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
|
||||
}
|
||||
|
||||
|
||||
@@ -47,6 +47,7 @@ struct RegState {
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case PREDClass: PREGs[Reg.Reg] = ssa; return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -59,6 +60,7 @@ struct RegState {
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
case PREDClass: return PREGs[Reg.Reg];
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
@@ -82,6 +84,7 @@ private:
|
||||
std::array<IR::NodeID, 32> FPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> GPRs = {};
|
||||
std::array<IR::NodeID, 32> FPRs = {};
|
||||
std::array<IR::NodeID, 32> PREGs = {};
|
||||
|
||||
fextl::unordered_map<uint32_t, IR::NodeID> Spills;
|
||||
};
|
||||
|
||||
@@ -7,6 +7,9 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/fextl/deque.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
@@ -15,16 +18,13 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
|
||||
// Flag bit flags
|
||||
#define FLAG_V (1U << 0)
|
||||
#define FLAG_C (1U << 1)
|
||||
#define FLAG_Z (1U << 2)
|
||||
#define FLAG_N (1U << 3)
|
||||
#define FLAG_A (1U << 4)
|
||||
#define FLAG_P (1U << 5)
|
||||
#define FLAG_P (1U << 4)
|
||||
#define FLAG_A (1U << 5)
|
||||
|
||||
#define FLAG_ZCV (FLAG_Z | FLAG_C | FLAG_V)
|
||||
#define FLAG_NZCV (FLAG_N | FLAG_ZCV)
|
||||
@@ -32,10 +32,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
struct FlagInfo {
|
||||
// If set, all following fields are zero, used for a quick exit.
|
||||
bool Trivial;
|
||||
|
||||
struct FlagInfoUnpacked {
|
||||
// Set of flags read by the instruction.
|
||||
unsigned Read;
|
||||
|
||||
@@ -46,15 +43,106 @@ struct FlagInfo {
|
||||
// eliminated.
|
||||
bool CanEliminate;
|
||||
|
||||
// If true, the opcode can be replaced with Replacement if its flag writes can
|
||||
// all be eliminated.
|
||||
bool CanReplace;
|
||||
// If set, the opcode can be replaced with Replacement if its flag writes can
|
||||
// all be eliminated, or ReplacementNoWrite if its register write can be
|
||||
// eliminated.
|
||||
IROps Replacement;
|
||||
|
||||
// If true, the opcode can be replaced with ReplacementNoWrite if its register
|
||||
// write is unused but its flags are still needed.
|
||||
bool CanReplaceWrite;
|
||||
IROps ReplacementNoWrite;
|
||||
|
||||
// Needs speical handling
|
||||
bool Special;
|
||||
};
|
||||
|
||||
struct FlagInfo {
|
||||
uint64_t Raw;
|
||||
|
||||
static constexpr struct FlagInfo Pack(struct FlagInfoUnpacked F) {
|
||||
uint64_t R = F.Read | (F.Write << 8) | (F.CanEliminate << 16) | (((uint64_t)F.Replacement) << 32) |
|
||||
((uint64_t)F.ReplacementNoWrite << 48) | (F.Special ? (1ull << 63) : 0);
|
||||
return {.Raw = R};
|
||||
}
|
||||
|
||||
bool Trivial() {
|
||||
return Raw == 0;
|
||||
}
|
||||
|
||||
unsigned Read() {
|
||||
return Bits(0, 8);
|
||||
}
|
||||
|
||||
unsigned Write() {
|
||||
return Bits(8, 8);
|
||||
}
|
||||
|
||||
bool CanEliminate() {
|
||||
return Bits(16, 1);
|
||||
}
|
||||
|
||||
bool Special() {
|
||||
return Bits(63, 1);
|
||||
}
|
||||
|
||||
IROps Replacement() {
|
||||
return (IROps)Bits(32, 16);
|
||||
}
|
||||
|
||||
IROps ReplacementNoWrite() {
|
||||
return (IROps)Bits(48, 16);
|
||||
}
|
||||
|
||||
private:
|
||||
unsigned Bits(unsigned Start, unsigned Count) {
|
||||
return (Raw >> Start) & ((1u << Count) - 1);
|
||||
}
|
||||
};
|
||||
|
||||
struct BlockInfo {
|
||||
fextl::vector<Ref> Predecessors;
|
||||
uint8_t Flags;
|
||||
bool InWorklist;
|
||||
};
|
||||
|
||||
struct ControlFlowGraph {
|
||||
fextl::unordered_map<uint32_t, BlockInfo> BlockMap;
|
||||
IRListView& IR;
|
||||
|
||||
void AddBlock(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
uint32_t ID = IR.GetID(Block).Value;
|
||||
|
||||
// Add the block with conservative flags and already in the worklist.
|
||||
auto Info = &BlockMap.emplace(ID, BlockInfo {{}, FLAG_ALL, true}).first->second;
|
||||
|
||||
// Add some initial capacity
|
||||
Info->Predecessors.reserve(2);
|
||||
|
||||
// Add to worklist
|
||||
Worklist.push_back(Block);
|
||||
}
|
||||
|
||||
BlockInfo* Get(uint32_t Block) {
|
||||
return &BlockMap.try_emplace(Block).first->second;
|
||||
}
|
||||
|
||||
BlockInfo* Get(Ref Block) {
|
||||
return Get(IR.GetID(Block).Value);
|
||||
}
|
||||
|
||||
BlockInfo* Get(OrderedNodeWrapper Block) {
|
||||
return Get(Block.ID().Value);
|
||||
}
|
||||
|
||||
void RecordEdge(Ref From, Ref To) {
|
||||
auto Info = Get(To);
|
||||
Info->Predecessors.push_back(From);
|
||||
}
|
||||
|
||||
void AddWorklist(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
auto Info = Get(Block);
|
||||
if (!Info->InWorklist) {
|
||||
Info->InWorklist = true;
|
||||
Worklist.push_front(Block);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass {
|
||||
@@ -66,10 +154,10 @@ private:
|
||||
unsigned FlagForReg(unsigned Reg);
|
||||
unsigned FlagsForCondClassType(CondClassType Cond);
|
||||
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagForReg(unsigned Reg) {
|
||||
return Reg == Core::CPUState::PF_AS_GREG ? FLAG_P : Reg == Core::CPUState::AF_AS_GREG ? FLAG_A : 0;
|
||||
void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode);
|
||||
CondClassType X86ToArmFloatCond(CondClassType X86);
|
||||
bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG);
|
||||
void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
|
||||
@@ -107,160 +195,192 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType C
|
||||
}
|
||||
}
|
||||
|
||||
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
constexpr FlagInfo ClassifyConst(IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_ANDWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_AND,
|
||||
};
|
||||
.ReplacementNoWrite = OP_TESTNZ,
|
||||
});
|
||||
|
||||
case OP_ADDWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADD,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_ADDNZCV,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_SUBWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_SUB,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_SUBNZCV,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_ADCWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADC,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_ADCNZCV,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_ADCZEROWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADCZERO,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_SBBWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_SBB,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_SBBNZCV,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_SHIFTFLAGS:
|
||||
// _ShiftFlags conditionally sets NZCV+PF, which we model here as a
|
||||
// read-modify-write. Logically, it also conditionally makes AF undefined,
|
||||
// which we model by omitting AF from both Read and Write sets (since
|
||||
// "cond ? AF : undef" may be optimized to "AF").
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_NZCV | FLAG_P,
|
||||
.Write = FLAG_NZCV | FLAG_P,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_ROTATEFLAGS:
|
||||
// _RotateFlags conditionally sets CV, again modeled as RMW.
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C | FLAG_V,
|
||||
.Write = FLAG_C | FLAG_V,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_RDRAND: return {.Write = FLAG_NZCV};
|
||||
case OP_RDRAND: return FlagInfo::Pack({.Write = FLAG_NZCV});
|
||||
|
||||
case OP_ADDNZCV:
|
||||
case OP_SUBNZCV:
|
||||
case OP_TESTNZ:
|
||||
case OP_FCMP:
|
||||
case OP_STORENZCV:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_AXFLAG:
|
||||
// Per the Arm spec, axflag reads Z/V/C but not N. It writes all flags.
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_ZCV,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_CMPPAIRZ:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_Z,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_CARRYINVERT:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_C,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_SETSMALLNZV:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_N | FLAG_Z | FLAG_V,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_LOADNZCV: return {.Read = FLAG_NZCV};
|
||||
case OP_LOADNZCV: return FlagInfo::Pack({.Read = FLAG_NZCV});
|
||||
|
||||
case OP_ADC:
|
||||
case OP_SBB: return {.Read = FLAG_C};
|
||||
case OP_ADCZERO:
|
||||
case OP_SBB: return FlagInfo::Pack({.Read = FLAG_C});
|
||||
|
||||
case OP_ADCNZCV:
|
||||
case OP_SBBNZCV:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_LOADPF: return FlagInfo::Pack({.Read = FLAG_P});
|
||||
case OP_LOADAF: return FlagInfo::Pack({.Read = FLAG_A});
|
||||
case OP_STOREPF: return FlagInfo::Pack({.Write = FLAG_P, .CanEliminate = true});
|
||||
case OP_STOREAF: return FlagInfo::Pack({.Write = FLAG_A, .CanEliminate = true});
|
||||
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTV:
|
||||
case OP_NZCVSELECTINCREMENT:
|
||||
case OP_NEG:
|
||||
case OP_CONDJUMP:
|
||||
case OP_CONDSUBNZCV:
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_RMIFNZCV:
|
||||
case OP_INVALIDATEFLAGS: return FlagInfo::Pack({.Special = true});
|
||||
default: return FlagInfo::Pack({});
|
||||
}
|
||||
}
|
||||
|
||||
constexpr auto FlagInfos = std::invoke([] {
|
||||
std::array<FlagInfo, OP_LAST> ret = {};
|
||||
|
||||
for (unsigned i = 0; i < OP_LAST; ++i) {
|
||||
ret[i] = ClassifyConst((IROps)i);
|
||||
}
|
||||
|
||||
return ret;
|
||||
});
|
||||
|
||||
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
FlagInfo Info = FlagInfos[IROp->Op];
|
||||
if (!Info.Special()) {
|
||||
return Info;
|
||||
}
|
||||
|
||||
switch (IROp->Op) {
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTINCREMENT: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NZCVSELECTV: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelectV>();
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NEG: {
|
||||
auto Op = IROp->CW<IR::IROp_Neg>();
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
if (!Op->FromNZCV) {
|
||||
break;
|
||||
return FlagInfo::Pack({});
|
||||
}
|
||||
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_CONDSUBNZCV:
|
||||
case OP_CONDADDNZCV: {
|
||||
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FlagsForCondClassType(Op->Cond),
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
case OP_RMIFNZCV: {
|
||||
@@ -271,10 +391,10 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
static_assert(FLAG_C == (1 << 1), "rmif mask lines up with our bits");
|
||||
static_assert(FLAG_V == (1 << 0), "rmif mask lines up with our bits");
|
||||
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = Op->Mask,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
case OP_INVALIDATEFLAGS: {
|
||||
@@ -309,39 +429,16 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
// The mental model of InvalidateFlags is writing undefined values to all
|
||||
// of the selected flags, allowing the write-after-write optimizations to
|
||||
// optimize invalidate-after-write for free.
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = Flags,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
case OP_LOADREGISTER: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadRegister>();
|
||||
if (Op->Class != GPRClass) {
|
||||
break;
|
||||
}
|
||||
|
||||
return {.Read = FlagForReg(Op->Reg)};
|
||||
default: LOGMAN_THROW_AA_FMT(false, "invalid special op"); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
case OP_STOREREGISTER: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreRegister>();
|
||||
if (Op->Class != GPRClass) {
|
||||
break;
|
||||
}
|
||||
|
||||
unsigned Flag = FlagForReg(Op->Reg);
|
||||
|
||||
return {
|
||||
.Write = Flag,
|
||||
.CanEliminate = Flag != 0,
|
||||
};
|
||||
}
|
||||
|
||||
default: break;
|
||||
}
|
||||
|
||||
return {.Trivial = true};
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// General purpose dead code elimination. Returns whether flag handling should
|
||||
@@ -385,83 +482,295 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref
|
||||
return true;
|
||||
}
|
||||
|
||||
CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) {
|
||||
// Table of x86 condition codes that map to arm64 condition codes, in the
|
||||
// sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm).
|
||||
//
|
||||
// E would be "equal or unordered", no condition code.
|
||||
// G would be "greater than or less than", no condition code.
|
||||
//
|
||||
// SF/OF conditions are trivial and therefore shouldn't actually be generated
|
||||
switch (X86) {
|
||||
case COND_UGE /* A */: return {COND_FGE} /* GE */;
|
||||
case COND_UGT /* AE */: return {COND_FGT} /* GT */;
|
||||
case COND_ULT /* B */: return {COND_SLT} /* LT */;
|
||||
case COND_ULE /* BE */: return {COND_SLE} /* LE */;
|
||||
case COND_SLE /* LE */: return {COND_SLE} /* LE */;
|
||||
default: return {COND_AL};
|
||||
}
|
||||
}
|
||||
|
||||
void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode) {
|
||||
// Skip past StoreRegisters at the end -- they don't touch flags.
|
||||
auto PrevWrap = CodeNode->Header.Previous;
|
||||
while (CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREREGISTER ||
|
||||
CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREPF || CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREAF) {
|
||||
PrevWrap = CurrentIR.GetNode(PrevWrap)->Header.Previous;
|
||||
}
|
||||
|
||||
auto Prev = CurrentIR.GetOp<IR::IROp_Header>(PrevWrap);
|
||||
if (Prev->Op == OP_AXFLAG) {
|
||||
// Pattern match a branch fed by AXFLAG.
|
||||
CondClassType ArmCond = X86ToArmFloatCond(Op->Cond);
|
||||
if (ArmCond == COND_AL) {
|
||||
return;
|
||||
}
|
||||
|
||||
Op->Cond = ArmCond;
|
||||
} else if (Prev->Op == OP_SUBNZCV) {
|
||||
// Pattern match a branch fed by a compare. We could also handle bit tests
|
||||
// here, but tbz/tbnz has a limited offset range which we don't have a way to
|
||||
// deal with yet. Let's hope that's not a big deal.
|
||||
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < OpSize::i32Bit)) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto SecondArg = CurrentIR.GetOp<IR::IROp_Header>(Prev->Args[1]);
|
||||
if (SecondArg->Op != OP_INLINECONSTANT || SecondArg->C<IR::IROp_InlineConstant>()->Constant != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We've matched. Fold the compare into branch.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CurrentIR.GetNode(Prev->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CurrentIR.GetNode(Prev->Args[1]));
|
||||
Op->FromNZCV = false;
|
||||
Op->CompareSize = Prev->Size;
|
||||
} else {
|
||||
return;
|
||||
}
|
||||
|
||||
// The compare/test/axflag sets flags but does not write registers. Flags are
|
||||
// dead after the jump. The jump does not read flags anymore. There is no
|
||||
// intervening instruction. Therefore the compare is dead.
|
||||
IREmit->Remove(CurrentIR.GetNode(PrevWrap));
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief This pass removes dead code locally.
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG) {
|
||||
uint32_t FlagsRead = FLAG_ALL;
|
||||
|
||||
// Reverse iteration is not yet working with the iterators
|
||||
auto BlockIROp = CurrentIR.GetOp<IR::IROp_CodeBlock>(Block);
|
||||
|
||||
// We grab these nodes this way so we can iterate easily
|
||||
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
|
||||
// Advance past EndBlock to get at the exit.
|
||||
--CodeLast;
|
||||
|
||||
// Initialize the FlagsRead mask according to the exit instruction.
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
FlagsRead = CFG.Get(ExitOp->Args[0])->Flags;
|
||||
}
|
||||
|
||||
// Iterate the block in reverse
|
||||
while (true) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
// Optimizing flags can cause earlier flag reads to become dead but dead
|
||||
// flag reads should not impede optimiation of earlier dead flag writes.
|
||||
// We must DCE as we go to ensure we converge in a single iteration.
|
||||
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
|
||||
// Optimiation algorithm: For each flag written...
|
||||
//
|
||||
// If the flag has a later read (per FlagsRead), remove the flag from
|
||||
// FlagsRead, since the reader is covered by this write.
|
||||
//
|
||||
// Else, there is no later read, so remove the flag write (if we can).
|
||||
// This is the active part of the optimization.
|
||||
//
|
||||
// Then, add each flag read to FlagsRead.
|
||||
//
|
||||
// This order is important: instructions that read-modify-write flags
|
||||
// (like adcs) first read flags, then write flags. Since we're iterating
|
||||
// the block backwards, that means we handle the write first.
|
||||
struct FlagInfo Info = Classify(IROp);
|
||||
|
||||
if (!Info.Trivial()) {
|
||||
bool Eliminated = false;
|
||||
|
||||
if ((FlagsRead & Info.Write()) == 0) {
|
||||
if ((Info.CanEliminate() || Info.Replacement()) && CodeNode->GetUses() == 0) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Eliminated = true;
|
||||
} else if (Info.Replacement()) {
|
||||
IROp->Op = Info.Replacement();
|
||||
}
|
||||
} else if (Info.ReplacementNoWrite() && CodeNode->GetUses() == 0) {
|
||||
IROp->Op = Info.ReplacementNoWrite();
|
||||
}
|
||||
|
||||
// If we don't care about the sign or carry, we can optimize testnz.
|
||||
// Carry is inverted between testz and testnz so we check that too. Note
|
||||
// this flag is outside of the if, since the TestNZ might result from
|
||||
// optimizing AndWithFlags, and we need to converge locally in a single
|
||||
// iteration.
|
||||
if (IROp->Op == OP_TESTNZ && IROp->Size < OpSize::i32Bit && !(FlagsRead & (FLAG_N | FLAG_C))) {
|
||||
IROp->Op = OP_TESTZ;
|
||||
}
|
||||
|
||||
FlagsRead &= ~Info.Write();
|
||||
|
||||
// If we eliminated the instruction, we eliminate its read too. This
|
||||
// check is required to ensure the pass converges locally in a single
|
||||
// iteration.
|
||||
if (!Eliminated) {
|
||||
FlagsRead |= Info.Read();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Iterate in reverse
|
||||
if (CodeLast == CodeBegin) {
|
||||
break;
|
||||
}
|
||||
--CodeLast;
|
||||
}
|
||||
|
||||
// For the purposes of global propagation, the content of our progress doesn't
|
||||
// matter -- only the difference in our final FlagsRead contributes to changes
|
||||
// in the predecessors.
|
||||
uint32_t OldFlagsRead = CFG.Get(Block)->Flags;
|
||||
CFG.Get(Block)->Flags = FlagsRead;
|
||||
return (OldFlagsRead != FlagsRead);
|
||||
}
|
||||
|
||||
void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG) {
|
||||
// Mapping for flags inside this pass.
|
||||
const uint8_t PARTIAL = 0;
|
||||
const uint8_t FULL = 1;
|
||||
|
||||
// Initialize conservatively: all blocks need full parity. This initialization
|
||||
// matters for proper handling of backedges.
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
CFG.Get(Block)->Flags = FULL;
|
||||
}
|
||||
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
bool Full = false;
|
||||
auto Predecessors = CFG.Get(Block)->Predecessors;
|
||||
|
||||
if (Predecessors.empty()) {
|
||||
// Conservatively assume there was full parity before the start block
|
||||
Full = true;
|
||||
} else {
|
||||
// If any predecessor needs full parity at the end, we need full parity.
|
||||
for (auto Pred : Predecessors) {
|
||||
Full |= (CFG.Get(Pred)->Flags == FULL);
|
||||
}
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) {
|
||||
if (IROp->Op == OP_STOREPF) {
|
||||
auto Op = IROp->CW<IR::IROp_StorePF>();
|
||||
auto Generator = CurrentIR.GetOp<IR::IROp_Header>(Op->Value);
|
||||
|
||||
// Determine if we only write 0/1 to the parity flag.
|
||||
Full = true;
|
||||
if (Generator->Op == OP_NZCVSELECT) {
|
||||
auto C0 = CurrentIR.GetOp<IR::IROp_Header>(Generator->Args[0]);
|
||||
auto C1 = CurrentIR.GetOp<IR::IROp_Header>(Generator->Args[1]);
|
||||
if (C0->Op == C1->Op && C0->Op == OP_INLINECONSTANT) {
|
||||
auto IC0 = CurrentIR.GetOp<IR::IROp_InlineConstant>(Generator->Args[0]);
|
||||
auto IC1 = CurrentIR.GetOp<IR::IROp_InlineConstant>(Generator->Args[1]);
|
||||
|
||||
// We need the full 8 if the constant has upper bits set.
|
||||
Full = (IC0->Constant | IC1->Constant) & ~1;
|
||||
}
|
||||
}
|
||||
} else if (IROp->Op == OP_PARITY && !Full) {
|
||||
// Eliminate parity calculations if it's only 1-bit.
|
||||
auto Parity = IROp->C<IROp_Parity>();
|
||||
Ref Value = CurrentIR.GetNode(Parity->Raw);
|
||||
|
||||
if (Parity->Invert) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Value = IREmit->_Xor(OpSize::i32Bit, Value, IREmit->_InlineConstant(1));
|
||||
}
|
||||
|
||||
IREmit->ReplaceUsesWithAfter(CodeNode, Value, CurrentIR.at(CodeNode));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
|
||||
// Record our final state for our successors to read.
|
||||
CFG.Get(Block)->Flags = Full ? FULL : PARTIAL;
|
||||
}
|
||||
}
|
||||
|
||||
void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::deque<Ref> Worklist;
|
||||
|
||||
ControlFlowGraph CFG {.IR = CurrentIR};
|
||||
|
||||
// Gather blocks
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
// We model all flags as read at the end of the block, since this pass is
|
||||
// presently purely local. Optimizing this requires global anslysis.
|
||||
uint32_t FlagsRead = FLAG_ALL;
|
||||
CFG.AddBlock(Worklist, BlockNode);
|
||||
}
|
||||
|
||||
// Reverse iteration is not yet working with the iterators
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
// Gather CFG
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto CodeLast = CurrentIR.at(BlockHeader->C<IROp_CodeBlock>()->Last);
|
||||
--CodeLast;
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
|
||||
// We grab these nodes this way so we can iterate easily
|
||||
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
|
||||
// Iterate the block in reverse
|
||||
while (1) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
// Optimizing flags can cause earlier flag reads to become dead but dead
|
||||
// flag reads should not impede optimiation of earlier dead flag writes.
|
||||
// We must DCE as we go to ensure we converge in a single iteration.
|
||||
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
|
||||
// Optimiation algorithm: For each flag written...
|
||||
//
|
||||
// If the flag has a later read (per FlagsRead), remove the flag from
|
||||
// FlagsRead, since the reader is covered by this write.
|
||||
//
|
||||
// Else, there is no later read, so remove the flag write (if we can).
|
||||
// This is the active part of the optimization.
|
||||
//
|
||||
// Then, add each flag read to FlagsRead.
|
||||
//
|
||||
// This order is important: instructions that read-modify-write flags
|
||||
// (like adcs) first read flags, then write flags. Since we're iterating
|
||||
// the block backwards, that means we handle the write first.
|
||||
struct FlagInfo Info = Classify(IROp);
|
||||
|
||||
if (!Info.Trivial) {
|
||||
bool Eliminated = false;
|
||||
|
||||
if ((FlagsRead & Info.Write) == 0) {
|
||||
if ((Info.CanEliminate || Info.CanReplace) && CodeNode->GetUses() == 0) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Eliminated = true;
|
||||
} else if (Info.CanReplace) {
|
||||
IROp->Op = Info.Replacement;
|
||||
}
|
||||
} else {
|
||||
FlagsRead &= ~Info.Write;
|
||||
|
||||
if (Info.CanReplaceWrite && CodeNode->GetUses() == 0) {
|
||||
IROp->Op = Info.ReplacementNoWrite;
|
||||
}
|
||||
}
|
||||
|
||||
// If we eliminated the instruction, we eliminate its read too. This
|
||||
// check is required to ensure the pass converges locally in a single
|
||||
// iteration.
|
||||
if (!Eliminated) {
|
||||
FlagsRead |= Info.Read;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Iterate in reverse
|
||||
if (CodeLast == CodeBegin) {
|
||||
break;
|
||||
}
|
||||
--CodeLast;
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->TrueBlock));
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->FalseBlock));
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(ExitOp->Args[0]));
|
||||
}
|
||||
}
|
||||
|
||||
// After processing a block, if we made progress, we must process its
|
||||
// predecessors to propagate globally. A block will be reprocessed only if
|
||||
// there is a loop backedge.
|
||||
for (; !Worklist.empty(); Worklist.pop_back()) {
|
||||
auto Block = Worklist.back();
|
||||
auto Info = CFG.Get(Block);
|
||||
Info->InWorklist = false;
|
||||
|
||||
if (ProcessBlock(IREmit, CurrentIR, Block, CFG)) {
|
||||
for (auto Pred : Info->Predecessors) {
|
||||
CFG.AddWorklist(Worklist, Pred);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fold compares into branches now that we're otherwise optimized. This needs
|
||||
// to run after eliminating carries etc and it needs the global flag metadata.
|
||||
// But it only needs to run once, we don't do it in the loop.
|
||||
for (auto [Block, _] : CurrentIR.GetBlocks()) {
|
||||
// Grab the jump
|
||||
auto BlockIROp = CurrentIR.GetOp<IR::IROp_CodeBlock>(Block);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
--CodeLast;
|
||||
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
uint32_t FlagsOut = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
|
||||
|
||||
if ((FlagsOut & FLAG_NZCV) == 0 && Op->FromNZCV) {
|
||||
FoldBranch(IREmit, CurrentIR, Op, ExitNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (CurrentIR.GetHeader()->ReadsParity) {
|
||||
OptimizeParity(IREmit, CurrentIR, CFG);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
|
||||
|
||||
@@ -28,13 +28,10 @@ namespace {
|
||||
uint32_t Available;
|
||||
uint32_t Count;
|
||||
|
||||
// If bit R of Allocated is 1, then RegToSSA[R] is the Old node
|
||||
// If bit R of Available is 0, then RegToSSA[R] is the Old node
|
||||
// currently allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to
|
||||
// clear this when freeing registers.
|
||||
Ref RegToSSA[32];
|
||||
|
||||
// Allocated base registers. Similar to ~Available except for pairs.
|
||||
uint32_t Allocated;
|
||||
};
|
||||
|
||||
IR::RegisterClassType GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
@@ -90,7 +87,7 @@ private:
|
||||
//
|
||||
// SSAToNewSSA tracks the current remapping. nullptr indicates no remapping.
|
||||
//
|
||||
// Since its indexed by Old nodes, SSAToNewSSA does not grow.
|
||||
// Since its indexed by Old nodes, SSAToNewSSA does not grow after allocation.
|
||||
fextl::vector<Ref> SSAToNewSSA;
|
||||
|
||||
// Inverse of SSAToNewSSA. Since it's indexed by new nodes, it grows.
|
||||
@@ -100,19 +97,27 @@ private:
|
||||
fextl::vector<PhysicalRegister> SSAToReg;
|
||||
|
||||
bool IsOld(Ref Node) {
|
||||
return IR->GetID(Node).Value < SSAToNewSSA.size();
|
||||
return IR->GetID(Node).Value < PreferredReg.size();
|
||||
};
|
||||
|
||||
// Return the New node (if it exists) for an Old node, else the Old node.
|
||||
Ref Map(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Pre-condition");
|
||||
|
||||
return SSAToNewSSA[IR->GetID(Old).Value] ?: Old;
|
||||
if (SSAToNewSSA.empty()) {
|
||||
return Old;
|
||||
} else {
|
||||
return SSAToNewSSA[IR->GetID(Old).Value] ?: Old;
|
||||
}
|
||||
};
|
||||
|
||||
// Return the Old node for a possibly-remapped node.
|
||||
Ref Unmap(Ref Node) {
|
||||
return NewSSAToSSA[IR->GetID(Node).Value] ?: Node;
|
||||
if (NewSSAToSSA.empty()) {
|
||||
return Node;
|
||||
} else {
|
||||
return NewSSAToSSA[IR->GetID(Node).Value] ?: Node;
|
||||
}
|
||||
};
|
||||
|
||||
// Record a remapping of Old to New.
|
||||
@@ -125,6 +130,10 @@ private:
|
||||
LOGMAN_THROW_A_FMT(NewID >= NewSSAToSSA.size(), "Brand new SSA def");
|
||||
NewSSAToSSA.resize(NewID + 1, 0);
|
||||
|
||||
if (SSAToNewSSA.empty()) {
|
||||
SSAToNewSSA.resize(PreferredReg.size(), nullptr);
|
||||
}
|
||||
|
||||
SSAToNewSSA[OldID] = New;
|
||||
NewSSAToSSA[NewID] = Old;
|
||||
|
||||
@@ -198,7 +207,7 @@ private:
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Allocated & GetRegBits(Reg)) && Class->RegToSSA[Reg.Reg] == Old;
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Old;
|
||||
};
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
@@ -206,10 +215,8 @@ private:
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
LOGMAN_THROW_AA_FMT((Class->Allocated & RegBits), "Register double-free");
|
||||
|
||||
Class->Available |= RegBits;
|
||||
Class->Allocated &= ~RegBits;
|
||||
};
|
||||
|
||||
bool HasSource(IROp_Header* I, Ref Old) {
|
||||
@@ -228,11 +235,14 @@ private:
|
||||
};
|
||||
|
||||
Ref DecodeSRANode(const IROp_Header* IROp, Ref Node) {
|
||||
if (IROp->Op == OP_LOADREGISTER) {
|
||||
if (IROp->Op == OP_LOADREGISTER || IROp->Op == OP_LOADPF || IROp->Op == OP_LOADAF) {
|
||||
return Node;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
return IR->GetNode(Op->Value);
|
||||
} else if (IROp->Op == OP_STOREPF || IROp->Op == OP_STOREAF) {
|
||||
const IROp_StorePF* Op = IROp->C<IR::IROp_StorePF>();
|
||||
return IR->GetNode(Op->Value);
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
@@ -242,6 +252,8 @@ private:
|
||||
RegisterClassType Class;
|
||||
uint8_t Reg;
|
||||
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
if (IROp->Op == OP_LOADREGISTER) {
|
||||
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
@@ -253,17 +265,15 @@ private:
|
||||
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Class == GPRClass || Class == FPRClass, "SRA classes");
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
if (Class == FPRClass) {
|
||||
return PhysicalRegister {FPRFixedClass, Reg};
|
||||
} else if (Reg == Core::CPUState::PF_AS_GREG) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
} else if (Reg == Core::CPUState::AF_AS_GREG) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
} else {
|
||||
return PhysicalRegister {GPRFixedClass, Reg};
|
||||
}
|
||||
@@ -280,8 +290,9 @@ private:
|
||||
Ref Candidate = nullptr;
|
||||
uint32_t BestDistance = UINT32_MAX;
|
||||
uint8_t BestReg = ~0;
|
||||
uint32_t Allocated = ((1u << Class->Count) - 1) & ~Class->Available;
|
||||
|
||||
foreach_bit(i, Class->Allocated) {
|
||||
foreach_bit(i, Allocated) {
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Old != nullptr, "Invariant3");
|
||||
@@ -347,10 +358,8 @@ private:
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Allocated & RegBits), "Precondition");
|
||||
|
||||
Class->Available &= ~RegBits;
|
||||
Class->Allocated |= (1u << Reg.Reg);
|
||||
Class->RegToSSA[Reg.Reg] = Unmap(Node);
|
||||
|
||||
if (Index >= SSAToReg.size()) {
|
||||
@@ -419,8 +428,7 @@ private:
|
||||
RegisterClassType ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClass* Class = &Classes[ClassType];
|
||||
|
||||
// Spill to make room in the register file. Free registers need not be
|
||||
// contiguous, we'll shuffle later.
|
||||
// Spill to make room in the register file.
|
||||
if (!Class->Available) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
SpillReg(Class, Pivot);
|
||||
@@ -458,9 +466,8 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
auto IR_ = IREmit->ViewIR();
|
||||
IR = &IR_;
|
||||
|
||||
// SSAToNewSSA, NewSSAToSSA allocated on first-use
|
||||
PreferredReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
SSAToNewSSA.resize(IR->GetSSACount(), nullptr);
|
||||
NewSSAToSSA.resize(IR->GetSSACount(), nullptr);
|
||||
SSAToReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
NextUses.resize(IR->GetSSACount(), 0);
|
||||
SpillSlotCount = 0;
|
||||
@@ -473,7 +480,6 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// At the start of each block, all registers are available.
|
||||
for (auto& Class : Classes) {
|
||||
Class.Available = (1u << Class.Count) - 1;
|
||||
Class.Allocated = 0;
|
||||
}
|
||||
|
||||
SourcesNextUses.clear();
|
||||
@@ -500,9 +506,9 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// of SourcesNextUses is consistent. The forward pass can then iterate
|
||||
// forwards and just flip the order.
|
||||
const uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
|
||||
for (int8_t i = NumArgs - 1; i >= 0; --i) {
|
||||
for (int i = NumArgs - 1; i >= 0; --i) {
|
||||
const auto& Arg = IROp->Args[i];
|
||||
if (IsValidArg(Arg)) {
|
||||
if (!Arg.IsInvalid()) {
|
||||
const uint32_t Index = Arg.ID().Value;
|
||||
|
||||
SourcesNextUses.push_back(NextUses[Index]);
|
||||
@@ -511,7 +517,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
// Record preferred registers for SRA. We also record the Node accessing
|
||||
// each register, used below. Since we initialized Class->Allocated = 0,
|
||||
// each register, used below. Since we initialized Class->Available,
|
||||
// RegToSSA is otherwise undefined so we can stash our temps there.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
@@ -564,7 +570,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
|
||||
if (Class->Allocated & (1u << Reg.Reg)) {
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
Ref Old = Class->RegToSSA[Reg.Reg];
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "RegToSSA invariant");
|
||||
@@ -612,21 +618,24 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
if (!IsValidArg(IROp->Args[s])) {
|
||||
if (IROp->Args[s].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
SourceIndex--;
|
||||
LOGMAN_THROW_AA_FMT(SourceIndex >= 0, "Consistent source count");
|
||||
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file");
|
||||
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
FreeReg(SSAToReg[IR->GetID(Map(Old)).Value]);
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
auto Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
|
||||
if (!Reg.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file");
|
||||
FreeReg(Reg);
|
||||
}
|
||||
}
|
||||
|
||||
NextUses[IR->GetID(Old).Value] = SourcesNextUses[SourceIndex];
|
||||
NextUses[IROp->Args[s].ID().Value] = SourcesNextUses[SourceIndex];
|
||||
}
|
||||
|
||||
// Assign destinations.
|
||||
@@ -635,11 +644,13 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
// Remap sources last, since AssignReg can shuffle.
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
Ref Remapped = SSAToNewSSA[IR->GetID(IR->GetNode(IROp->Args[s])).Value];
|
||||
if (!SSAToNewSSA.empty()) {
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
Ref Remapped = SSAToNewSSA[IROp->Args[s].ID().Value];
|
||||
|
||||
if (Remapped != nullptr) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, s, Remapped);
|
||||
if (Remapped != nullptr) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, s, Remapped);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -3,9 +3,10 @@
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -146,17 +147,18 @@ private:
|
||||
|
||||
class X87StackOptimization final : public Pass {
|
||||
public:
|
||||
X87StackOptimization() {
|
||||
X87StackOptimization(const FEXCore::HostFeatures& Features)
|
||||
: Features(Features) {
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
ReducedPrecisionMode = ReducedPrecision;
|
||||
}
|
||||
void Run(IREmitter* Emit) override;
|
||||
|
||||
private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
bool ReducedPrecisionMode;
|
||||
|
||||
// Helpers
|
||||
std::tuple<Ref, Ref> SplitF64SigExp(Ref Node);
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
|
||||
// Handles a Unary operation.
|
||||
@@ -284,7 +286,8 @@ inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
|
||||
|
||||
inline Ref X87StackOptimization::GetTopWithCache_Slow() {
|
||||
if (!TopOffsetCache[0]) {
|
||||
TopOffsetCache[0] = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
TopOffsetCache[0] =
|
||||
IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
return TopOffsetCache[0];
|
||||
}
|
||||
@@ -306,31 +309,32 @@ inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
|
||||
|
||||
|
||||
inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
|
||||
IREmit->_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
InvalidateTopOffsetCache();
|
||||
TopOffsetCache[0] = Value;
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::SetX87ValidTag(Ref Value, bool Valid) {
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), Value);
|
||||
Ref NewAbridgedFTW = Valid ? IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RegMask) : IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
|
||||
IREmit->_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetX87ValidTag_Slow(uint8_t Offset) {
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, AbridgedFTW, GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
|
||||
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? 8 : 16, MMBaseOffset(), 16, FPRClass);
|
||||
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
|
||||
MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset, bool SetValid) {
|
||||
OrderedNode* TopOffset = GetOffsetTopWithCache_Slow(Offset);
|
||||
// store
|
||||
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? 8 : 16, MMBaseOffset(), 16, FPRClass);
|
||||
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, MMBaseOffset(), 16, FPRClass);
|
||||
// mark it valid
|
||||
// In some cases we might already know it has been previously set as valid so we don't need to do it again
|
||||
if (SetValid) {
|
||||
@@ -379,7 +383,7 @@ void X87StackOptimization::HandleUnop(IROps Op64, bool VFOp64, IROps Op80) {
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
if (VFOp64) {
|
||||
DeriveOp(Value, Op64, IREmit->_VFSqrt(8, 8, St0));
|
||||
DeriveOp(Value, Op64, IREmit->_VFSqrt(OpSize::i64Bit, OpSize::i64Bit, St0));
|
||||
} else {
|
||||
DeriveOp(Value, Op64, IREmit->_F64SIN(St0));
|
||||
}
|
||||
@@ -399,10 +403,10 @@ void X87StackOptimization::HandleBinopValue(IROps Op64, bool VFOp64, IROps Op80,
|
||||
Ref Node = {};
|
||||
if (ReducedPrecisionMode) {
|
||||
if (Reverse) {
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(8, 8, ValueNode, StackNode));
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(OpSize::i64Bit, OpSize::i64Bit, ValueNode, StackNode));
|
||||
} else {
|
||||
if (VFOp64) {
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(8, 8, StackNode, ValueNode));
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(OpSize::i64Bit, OpSize::i64Bit, StackNode, ValueNode));
|
||||
} else {
|
||||
DeriveOp(Node, Op64, IREmit->_F64FPREM(StackNode, ValueNode));
|
||||
}
|
||||
@@ -476,13 +480,14 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
}
|
||||
Ref TopIndex = GetOffsetTopWithCache_Slow(i);
|
||||
if (Valid == StackSlot::VALID) {
|
||||
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? 8 : 16, MMBaseOffset(), 16, FPRClass);
|
||||
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
|
||||
MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
}
|
||||
{ // Set valid tags
|
||||
uint8_t Mask = StackData.getValidMask();
|
||||
if (Mask == 0xff) {
|
||||
IREmit->_StoreContext(1, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else if (Mask != 0) {
|
||||
if (std::popcount(Mask) == 1) {
|
||||
uint8_t BitIdx = __builtin_ctz(Mask);
|
||||
@@ -491,16 +496,16 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
// perform a rotate right on mask by top
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref NewAbridgedFTW = IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
|
||||
IREmit->_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
}
|
||||
}
|
||||
{ // Set invalid tags
|
||||
uint8_t Mask = StackData.getInvalidMask();
|
||||
if (Mask == 0xff) {
|
||||
IREmit->_StoreContext(1, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else if (Mask != 0) {
|
||||
if (std::popcount(Mask)) {
|
||||
uint8_t BitIdx = __builtin_ctz(Mask);
|
||||
@@ -509,29 +514,15 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
// Same rotate right as above but this time on the invalid mask
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref NewAbridgedFTW = IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
|
||||
IREmit->_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
}
|
||||
}
|
||||
return TopValue;
|
||||
}
|
||||
|
||||
std::tuple<Ref, Ref> X87StackOptimization::SplitF64SigExp(Ref Node) {
|
||||
Ref Gpr = IREmit->_VExtractToGPR(8, 8, Node, 0);
|
||||
|
||||
Ref Exp = IREmit->_And(OpSize::i64Bit, Gpr, GetConstant(0x7ff0000000000000LL));
|
||||
Exp = IREmit->_Lshr(OpSize::i64Bit, Exp, GetConstant(52));
|
||||
Exp = IREmit->_Sub(OpSize::i64Bit, Exp, GetConstant(1023));
|
||||
Exp = IREmit->_Float_FromGPR_S(8, 8, Exp);
|
||||
Ref Sig = IREmit->_And(OpSize::i64Bit, Gpr, GetConstant(0x800fffffffffffffLL));
|
||||
Sig = IREmit->_Or(OpSize::i64Bit, Sig, GetConstant(0x3ff0000000000000LL));
|
||||
Sig = IREmit->_VCastFromGPR(8, 8, Sig);
|
||||
|
||||
return std::tuple {Exp, Sig};
|
||||
}
|
||||
|
||||
void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::x87StackOpt");
|
||||
|
||||
@@ -644,7 +635,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_F80SINSTACK: {
|
||||
HandleUnop(OP_F64SIN, false, OP_F80SIN);
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -663,9 +653,9 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
HandleUnop(OP_F64TAN, false, OP_F80TAN);
|
||||
Ref OneConst {};
|
||||
if (ReducedPrecisionMode) {
|
||||
OneConst = IREmit->_VCastFromGPR(8, 8, GetConstant(0x3FF0000000000000));
|
||||
OneConst = IREmit->_VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, GetConstant(0x3FF0000000000000));
|
||||
} else {
|
||||
OneConst = IREmit->_LoadNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
OneConst = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
}
|
||||
|
||||
if (SlowPath) {
|
||||
@@ -724,7 +714,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
}
|
||||
} else { // invalidate all
|
||||
if (SlowPath) {
|
||||
IREmit->_StoreContext(1, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else {
|
||||
for (size_t i = 0; i < StackData.size; i++) {
|
||||
StackData.setTagInvalid(i);
|
||||
@@ -744,7 +734,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
} else {
|
||||
auto* SourceNode = CurrentIR.GetNode(Op->X80Src);
|
||||
auto* OriginalNode = CurrentIR.GetNode(Op->OriginalValue);
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, SizeToOpSize(Op->LoadSize), Op->Float});
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, Op->LoadSize, Op->Float});
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -810,33 +800,39 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
} else {
|
||||
if (ReducedPrecisionMode) {
|
||||
switch (Op->StoreSize) {
|
||||
case 4: {
|
||||
StackNode = IREmit->_Float_FToF(4, 8, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, 4, AddrNode, StackNode);
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i32Bit, AddrNode, StackNode);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
IREmit->_StoreMem(FPRClass, 8, AddrNode, StackNode);
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
break;
|
||||
}
|
||||
case 10: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, 8);
|
||||
IREmit->_StoreMem(FPRClass, 8, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(16, 8, StackNode, 1);
|
||||
IREmit->_StoreMem(GPRClass, 2, Upper, AddrNode, GetConstant(8), 8, MEM_OFFSET_SXTX, 1);
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, GetConstant(8), OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
} else {
|
||||
if (Op->StoreSize != 10) { // if it's not 80bits then convert
|
||||
if (Op->StoreSize != OpSize::f80Bit) { // if it's not 80bits then convert
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
}
|
||||
if (Op->StoreSize == 10) { // Part of code from StoreResult_WithOpSize()
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, 8, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(16, 8, StackNode, 1);
|
||||
auto DestAddr = IREmit->_Add(OpSize::i64Bit, AddrNode, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, 2, DestAddr, Upper, 8);
|
||||
if (Op->StoreSize == OpSize::f80Bit) { // Part of code from StoreResult_WithOpSize()
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
auto PReg = IREmit->InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
IREmit->_StoreMemPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, PReg, AddrNode);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto DestAddr = IREmit->_Add(OpSize::i64Bit, AddrNode, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, DestAddr, Upper, OpSize::i64Bit);
|
||||
}
|
||||
} else {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, AddrNode, StackNode);
|
||||
}
|
||||
@@ -887,13 +883,10 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// of a value
|
||||
Ref ResultNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
ResultNode = IREmit->_VFNeg(8, 8, Value);
|
||||
ResultNode = IREmit->_VFNeg(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
Ref Low = GetConstant(0);
|
||||
Ref High = GetConstant(0b1'000'0000'0000'0000ULL);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(16, 8, Low);
|
||||
HelperNode = IREmit->_VInsGPR(16, 8, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VXor(16, 1, Value, HelperNode);
|
||||
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
|
||||
ResultNode = IREmit->_VXor(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -904,14 +897,11 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref ResultNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
ResultNode = IREmit->_VFAbs(8, 8, Value);
|
||||
ResultNode = IREmit->_VFAbs(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
// Intermediate insts
|
||||
Ref Low = GetConstant(~0ULL);
|
||||
Ref High = GetConstant(0b0'111'1111'1111'1111ULL);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(16, 8, Low);
|
||||
HelperNode = IREmit->_VInsGPR(16, 8, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VAnd(16, 1, Value, HelperNode);
|
||||
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
|
||||
ResultNode = IREmit->_VAndn(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -925,7 +915,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref CmpNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
CmpNode = IREmit->_FCmp(8, StackValue1, StackValue2);
|
||||
CmpNode = IREmit->_FCmp(OpSize::i64Bit, StackValue1, StackValue2);
|
||||
} else {
|
||||
CmpNode = IREmit->_F80Cmp(StackValue1, StackValue2);
|
||||
}
|
||||
@@ -937,11 +927,11 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
const auto* Op = IROp->C<IROp_F80StackTest>();
|
||||
auto Offset = Op->SrcStack;
|
||||
auto StackNode = LoadStackValue(Offset);
|
||||
Ref ZeroConst = IREmit->_VCastFromGPR(ReducedPrecisionMode ? 8 : 16, 8, GetConstant(0));
|
||||
Ref ZeroConst = IREmit->_VCastFromGPR(ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, OpSize::i64Bit, GetConstant(0));
|
||||
|
||||
Ref CmpNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
CmpNode = IREmit->_FCmp(8, StackNode, ZeroConst);
|
||||
CmpNode = IREmit->_FCmp(OpSize::i64Bit, StackNode, ZeroConst);
|
||||
} else {
|
||||
CmpNode = IREmit->_F80Cmp(StackNode, ZeroConst);
|
||||
}
|
||||
@@ -957,7 +947,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref CmpNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
CmpNode = IREmit->_FCmp(8, StackNode, Value);
|
||||
CmpNode = IREmit->_FCmp(OpSize::i64Bit, StackNode, Value);
|
||||
} else {
|
||||
CmpNode = IREmit->_F80Cmp(StackNode, Value);
|
||||
}
|
||||
@@ -965,30 +955,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_F80XTRACTSTACK: {
|
||||
Ref St0 = LoadStackValue();
|
||||
|
||||
Ref Exp {};
|
||||
Ref Sig {};
|
||||
if (ReducedPrecisionMode) {
|
||||
std::tie(Exp, Sig) = SplitF64SigExp(St0);
|
||||
} else {
|
||||
Exp = IREmit->_F80XTRACT_EXP(St0);
|
||||
Sig = IREmit->_F80XTRACT_SIG(St0);
|
||||
}
|
||||
|
||||
if (SlowPath) {
|
||||
// Write exp to top, update top for a push and set sig at new top.
|
||||
StoreStackValueAtOffset_Slow(Exp, 0, false);
|
||||
UpdateTopForPush_Slow();
|
||||
StoreStackValueAtOffset_Slow(Sig);
|
||||
} else {
|
||||
StackData.setTop(StackMemberInfo {Exp});
|
||||
StackData.push(StackMemberInfo {Sig});
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SYNCSTACKTOSLOW: {
|
||||
// This synchronizes stack values but doesn't necessarily moves us off the FastPath!
|
||||
Ref NewTop = SynchronizeStackValues();
|
||||
@@ -1024,7 +990,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref Value {};
|
||||
if (ReducedPrecisionMode) {
|
||||
Value = IREmit->_Vector_FToI(8, 8, St0, Round_Host);
|
||||
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, Round_Host);
|
||||
} else {
|
||||
Value = IREmit->_F80Round(St0);
|
||||
}
|
||||
@@ -1040,7 +1006,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
Ref Value1 = LoadStackValue(StackOffset1);
|
||||
Ref Value2 = LoadStackValue(StackOffset2);
|
||||
|
||||
Ref StackNode = IREmit->_VBSL(16, CurrentIR.GetNode(Op->VectorMask), Value1, Value2);
|
||||
Ref StackNode = IREmit->_VBSL(OpSize::i128Bit, CurrentIR.GetNode(Op->VectorMask), Value1, Value2);
|
||||
StoreStackValue(StackNode, 0, StackOffset1 && StackOffset2);
|
||||
break;
|
||||
}
|
||||
@@ -1061,7 +1027,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass() {
|
||||
return fextl::make_unique<X87StackOptimization>();
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -19,24 +19,11 @@
|
||||
#include <sys/user.h>
|
||||
#endif
|
||||
|
||||
#ifdef ENABLE_JEMALLOC
|
||||
#include <jemalloc/jemalloc.h>
|
||||
#endif
|
||||
#include <errno.h>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
extern "C" {
|
||||
typedef void* (*mmap_hook_type)(void* addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
typedef int (*munmap_hook_type)(void* addr, size_t length);
|
||||
|
||||
#ifdef ENABLE_JEMALLOC
|
||||
extern mmap_hook_type je___mmap_hook;
|
||||
extern munmap_hook_type je___munmap_hook;
|
||||
#endif
|
||||
}
|
||||
|
||||
namespace fextl::pmr {
|
||||
static fextl::pmr::default_resource FEXDefaultResource;
|
||||
std::pmr::memory_resource* get_default_resource() {
|
||||
@@ -127,19 +114,15 @@ void ReenableSBRKAllocations(void* Ptr) {
|
||||
#pragma GCC diagnostic ignored "-Wdeprecated-declarations"
|
||||
void SetupHooks() {
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocator();
|
||||
#ifdef ENABLE_JEMALLOC
|
||||
je___mmap_hook = FEX_mmap;
|
||||
je___munmap_hook = FEX_munmap;
|
||||
#endif
|
||||
SetJemallocMmapHook(FEX_mmap);
|
||||
SetJemallocMunmapHook(FEX_munmap);
|
||||
FEXCore::Allocator::mmap = FEX_mmap;
|
||||
FEXCore::Allocator::munmap = FEX_munmap;
|
||||
}
|
||||
|
||||
void ClearHooks() {
|
||||
#ifdef ENABLE_JEMALLOC
|
||||
je___mmap_hook = ::mmap;
|
||||
je___munmap_hook = ::munmap;
|
||||
#endif
|
||||
SetJemallocMmapHook(::mmap);
|
||||
SetJemallocMunmapHook(::munmap);
|
||||
FEXCore::Allocator::mmap = ::mmap;
|
||||
FEXCore::Allocator::munmap = ::munmap;
|
||||
|
||||
@@ -239,7 +222,7 @@ fextl::vector<MemoryRegion> CollectMemoryGaps(uintptr_t Begin, uintptr_t End, in
|
||||
|
||||
// Parse mapped region in the format "fffff7cc3000-fffff7cc4000 r--p ..."
|
||||
{
|
||||
uintptr_t RegionBegin;
|
||||
uintptr_t RegionBegin {};
|
||||
auto result = std::from_chars(Cursor, line_end, RegionBegin, 16);
|
||||
LogMan::Throw::AFmt(result.ec == std::errc {} && *result.ptr == '-', "Unexpected line format");
|
||||
Cursor = result.ptr + 1;
|
||||
|
||||
Loaded 100 of 1328 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user