mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 23:00:17 +02:00
Compare commits
951
Commits
FEX-2209
...
FEX-2301_1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e9e88968d7 | ||
|
|
70d4a436cf | ||
|
|
ec55ecdb31 | ||
|
|
12b866c276 | ||
|
|
1038ba7060 | ||
|
|
9bde513161 | ||
|
|
2e3f77c43f | ||
|
|
0d8de6463b | ||
|
|
98349ee485 | ||
|
|
e486833d75 | ||
|
|
8a38999c7a | ||
|
|
2d0b61fd9d | ||
|
|
ffd9bb547d | ||
|
|
7a6ef8821f | ||
|
|
64387bf00d | ||
|
|
8cc7e55394 | ||
|
|
874c1da1b5 | ||
|
|
3904a5264f | ||
|
|
b95c1719c3 | ||
|
|
dfb3f31453 | ||
|
|
c6297edac0 | ||
|
|
8031f76642 | ||
|
|
4786ddc44c | ||
|
|
ae8a5fa98d | ||
|
|
6914598f9a | ||
|
|
450aedc8b6 | ||
|
|
1e221210a2 | ||
|
|
58ec2b2d7f | ||
|
|
438adf2f45 | ||
|
|
9707e9a4df | ||
|
|
9b8c92e275 | ||
|
|
6caf764b7c | ||
|
|
faa81f241b | ||
|
|
769c548ba4 | ||
|
|
dae1676e4a | ||
|
|
74526d1f02 | ||
|
|
8c005db81c | ||
|
|
32b70c8590 | ||
|
|
bdda14eb75 | ||
|
|
b8c0b0c267 | ||
|
|
ad7dc6ca0a | ||
|
|
64cd377e37 | ||
|
|
45d7564716 | ||
|
|
c585bae85d | ||
|
|
d07383fa73 | ||
|
|
e1cdcf0651 | ||
|
|
138f1fc844 | ||
|
|
ae69aa9a81 | ||
|
|
6341ac6814 | ||
|
|
6bc1c3fc30 | ||
|
|
d91f2ed6b0 | ||
|
|
7b30a241c0 | ||
|
|
aaf8e3757d | ||
|
|
d9c49c4ce1 | ||
|
|
4560c5b73c | ||
|
|
9d05c8a67b | ||
|
|
be9578551a | ||
|
|
4a884802f8 | ||
|
|
75a01ed2b6 | ||
|
|
3ebe141032 | ||
|
|
31e332bd61 | ||
|
|
764324d557 | ||
|
|
f37938576d | ||
|
|
16969fcdad | ||
|
|
0dfe141d70 | ||
|
|
2a7795fe2c | ||
|
|
b00b41b8fa | ||
|
|
39396789b1 | ||
|
|
38a2886a59 | ||
|
|
bd8e1a80f6 | ||
|
|
b7358b4926 | ||
|
|
0e0f3f9290 | ||
|
|
dfa113dcdb | ||
|
|
bf7d0f7ed9 | ||
|
|
82adc2f931 | ||
|
|
94cb2ddae7 | ||
|
|
0496f8d5c1 | ||
|
|
4a3af8d7f9 | ||
|
|
37a9588855 | ||
|
|
24f72b5f30 | ||
|
|
d927c4a903 | ||
|
|
12afe95602 | ||
|
|
7e715b9e04 | ||
|
|
9d58514f57 | ||
|
|
4f9402e5dd | ||
|
|
789093d158 | ||
|
|
33e8f21ac7 | ||
|
|
d672528e62 | ||
|
|
d7c959090d | ||
|
|
0a4846a524 | ||
|
|
cecda7bbb6 | ||
|
|
2d9cb65d5c | ||
|
|
e21002e0d7 | ||
|
|
ce351282f2 | ||
|
|
ab41856328 | ||
|
|
345e9b97bf | ||
|
|
0c651dd5f8 | ||
|
|
f850a02d3f | ||
|
|
983b53a0c2 | ||
|
|
10a6b5794b | ||
|
|
2c5aceb9b6 | ||
|
|
1668db046d | ||
|
|
825e921940 | ||
|
|
515b3e485b | ||
|
|
c06f0b7cb3 | ||
|
|
4aed60ee3d | ||
|
|
da9f7ec31f | ||
|
|
8ec932fc4c | ||
|
|
6f35a23161 | ||
|
|
bd70af9724 | ||
|
|
bdba062f72 | ||
|
|
60a2fb163c | ||
|
|
d591b1ed8c | ||
|
|
3bae4a225c | ||
|
|
d0cb329608 | ||
|
|
bb80e7d45c | ||
|
|
72a3b18279 | ||
|
|
109ed7d112 | ||
|
|
133a644231 | ||
|
|
666f8bfbd9 | ||
|
|
1800451251 | ||
|
|
1d9218224f | ||
|
|
7931bd1004 | ||
|
|
58978dd047 | ||
|
|
25fb243ac7 | ||
|
|
84f1e7ad4c | ||
|
|
3fb5835453 | ||
|
|
bcb6726b22 | ||
|
|
2bed562eb6 | ||
|
|
bae7209224 | ||
|
|
b53f8944ac | ||
|
|
03a061339a | ||
|
|
0d7c086b69 | ||
|
|
b958fa39a5 | ||
|
|
2b6a020c4c | ||
|
|
6e733bfc22 | ||
|
|
873d63002a | ||
|
|
bb6a0f39f5 | ||
|
|
392e6ae424 | ||
|
|
01d22849cf | ||
|
|
0537f2d014 | ||
|
|
f57debeb29 | ||
|
|
4ac031df59 | ||
|
|
78b53bfa49 | ||
|
|
c53fb7d697 | ||
|
|
a1a52450cb | ||
|
|
fabf453046 | ||
|
|
68916ae2d9 | ||
|
|
bf56b7b2da | ||
|
|
905eb015c0 | ||
|
|
858f13e76a | ||
|
|
169d7bbf50 | ||
|
|
31c8d4acac | ||
|
|
8291e600fa | ||
|
|
b26e4109fa | ||
|
|
dcc218a168 | ||
|
|
49b9b18b4a | ||
|
|
ad3bf189c0 | ||
|
|
47b21fa758 | ||
|
|
b6e82965df | ||
|
|
8dc8785340 | ||
|
|
c710ab60b0 | ||
|
|
c86ba7646c | ||
|
|
c1e301a5ed | ||
|
|
cad0dc6848 | ||
|
|
0ebb15c732 | ||
|
|
37c743b616 | ||
|
|
c810ae4018 | ||
|
|
d3481c8271 | ||
|
|
7c1e152441 | ||
|
|
f11ac8674d | ||
|
|
1fecf89bfc | ||
|
|
21ad0fa334 | ||
|
|
3429815103 | ||
|
|
559ff1582e | ||
|
|
2e973ae079 | ||
|
|
1ab4471ef9 | ||
|
|
40e073c8b2 | ||
|
|
58fab721b3 | ||
|
|
8fac21e43f | ||
|
|
d9a1e97bc1 | ||
|
|
848f1a2f78 | ||
|
|
7b8a46d934 | ||
|
|
9a8852f9b6 | ||
|
|
65e8bf9d72 | ||
|
|
344ec33ba5 | ||
|
|
5dc7dfacb3 | ||
|
|
122aa8a69a | ||
|
|
8ce6c08152 | ||
|
|
1beb791d52 | ||
|
|
27c0d4a9f5 | ||
|
|
dd4ba7562f | ||
|
|
bd9d8e8fe5 | ||
|
|
c7ac204322 | ||
|
|
4c013c867f | ||
|
|
5e634fcbc9 | ||
|
|
91c00d2cb6 | ||
|
|
e985dcdb22 | ||
|
|
ee9778480d | ||
|
|
048daa4579 | ||
|
|
6ae8a1e55f | ||
|
|
dc2eaf6511 | ||
|
|
0e233a96f0 | ||
|
|
ba5fafcd7f | ||
|
|
b12503fe32 | ||
|
|
aa63c7b94d | ||
|
|
cccbb7f595 | ||
|
|
ce12ed60ae | ||
|
|
d7eab5f787 | ||
|
|
21537a3636 | ||
|
|
588a2611a7 | ||
|
|
2895a09101 | ||
|
|
5c8d40d9be | ||
|
|
b4079cfea3 | ||
|
|
2b5570a910 | ||
|
|
6bb0c5b24c | ||
|
|
bc31f98f16 | ||
|
|
4b891d6147 | ||
|
|
58c3e20bd1 | ||
|
|
d5f3a091d0 | ||
|
|
a14e03f35d | ||
|
|
5c1789952e | ||
|
|
f5809f24f7 | ||
|
|
122a9114a3 | ||
|
|
d8f226b460 | ||
|
|
7171c5ae39 | ||
|
|
798a78534a | ||
|
|
ae4a04b560 | ||
|
|
1971c8d505 | ||
|
|
1ca356371d | ||
|
|
27ea6096a2 | ||
|
|
2244dd9847 | ||
|
|
ca2f4bd468 | ||
|
|
6b5c94be23 | ||
|
|
779dc48d8d | ||
|
|
4b2164768f | ||
|
|
90828aeb11 | ||
|
|
fe7c6da1e2 | ||
|
|
f3d0fa6f60 | ||
|
|
a9ad0d081c | ||
|
|
54885bec32 | ||
|
|
e8aa79bea9 | ||
|
|
60a45615df | ||
|
|
8a961bfcc5 | ||
|
|
d6ab7a4f97 | ||
|
|
7114fb3293 | ||
|
|
8a87aff730 | ||
|
|
ded257c92f | ||
|
|
c5b4719793 | ||
|
|
0f6201108f | ||
|
|
b589dce7f5 | ||
|
|
9de5840f7a | ||
|
|
c98fffd33d | ||
|
|
de3777cc78 | ||
|
|
dd640e7a3d | ||
|
|
d53ddb73bf | ||
|
|
25e9333abb | ||
|
|
85766dd074 | ||
|
|
40bab6b58e | ||
|
|
b0e0a2a165 | ||
|
|
0efcb912b5 | ||
|
|
9d0cc58737 | ||
|
|
b671ed57ef | ||
|
|
a2a44d188a | ||
|
|
364064536b | ||
|
|
98a454169d | ||
|
|
6d44370f28 | ||
|
|
273e2977a8 | ||
|
|
7264b07d4f | ||
|
|
92351e7f33 | ||
|
|
757602bb1e | ||
|
|
6aaffec67f | ||
|
|
6f474cedd3 | ||
|
|
d9176114c5 | ||
|
|
287cee5b41 | ||
|
|
a90067fb1e | ||
|
|
90d23098db | ||
|
|
384a09bbf1 | ||
|
|
2fd29c47d4 | ||
|
|
e8aa8d89ec | ||
|
|
1bc013d5f0 | ||
|
|
469ff91311 | ||
|
|
ef14c411ce | ||
|
|
c2c5d176e4 | ||
|
|
2328430d2e | ||
|
|
a07a533640 | ||
|
|
8c59e3e9e2 | ||
|
|
fed861fa6b | ||
|
|
ce9969ee8f | ||
|
|
9330ca41ea | ||
|
|
eefcea49f4 | ||
|
|
ed1b060494 | ||
|
|
6db165e24a | ||
|
|
58d20f199e | ||
|
|
437ab47ae7 | ||
|
|
d6b137e6b7 | ||
|
|
e1de89af79 | ||
|
|
42d24c21e1 | ||
|
|
3590f090c7 | ||
|
|
b7e177c11c | ||
|
|
92f92ddbbe | ||
|
|
f8d851b9b5 | ||
|
|
1689742e96 | ||
|
|
462b6b8c1c | ||
|
|
db90390179 | ||
|
|
e15fa66225 | ||
|
|
04a1fa6dc2 | ||
|
|
f5a337a142 | ||
|
|
2b9d0314ce | ||
|
|
293734408c | ||
|
|
03fbb923b3 | ||
|
|
39f0c8542f | ||
|
|
6877d5b3ec | ||
|
|
a6c30b35dc | ||
|
|
a57f3a6264 | ||
|
|
fa0ff71ddf | ||
|
|
c91ccf2cbe | ||
|
|
41df5f816d | ||
|
|
573896d0b7 | ||
|
|
36a6264571 | ||
|
|
12f01bc93a | ||
|
|
777b2c7966 | ||
|
|
f0141f124d | ||
|
|
283b178285 | ||
|
|
d3a5eef08a | ||
|
|
1bac33ff44 | ||
|
|
327a6f52fd | ||
|
|
4f313f5d40 | ||
|
|
b42b4e03a4 | ||
|
|
ab14375a03 | ||
|
|
ace90aac95 | ||
|
|
c4c93f5bfe | ||
|
|
3504ba068e | ||
|
|
88b88c9cd3 | ||
|
|
e99928990e | ||
|
|
6733f83471 | ||
|
|
a14cce27a4 | ||
|
|
04d5b53389 | ||
|
|
a6b0181cd4 | ||
|
|
3afd5691a4 | ||
|
|
82ad26307c | ||
|
|
dc9737a394 | ||
|
|
eaef06d14e | ||
|
|
2123868a42 | ||
|
|
b891999a7f | ||
|
|
a53fd07bda | ||
|
|
8f213b75be | ||
|
|
b73aeb8902 | ||
|
|
d642c1a646 | ||
|
|
d965ae03c4 | ||
|
|
e42de0b645 | ||
|
|
9ef5247dd7 | ||
|
|
25428cb28c | ||
|
|
2125949d6d | ||
|
|
7ac21e794d | ||
|
|
4aa0f3d0a4 | ||
|
|
740c983f65 | ||
|
|
83bccc0032 | ||
|
|
a98920d4e4 | ||
|
|
d1ab636df1 | ||
|
|
26b629833e | ||
|
|
95964f8dd8 | ||
|
|
94ae2e3a9c | ||
|
|
eae33b0c50 | ||
|
|
e9aa368a62 | ||
|
|
5a37786da7 | ||
|
|
5ac44baa2c | ||
|
|
4cf3805950 | ||
|
|
1f5a1826a6 | ||
|
|
bf86df7a66 | ||
|
|
6a63ae2d9c | ||
|
|
a7a1e2abd3 | ||
|
|
3322f8b890 | ||
|
|
047ae13c98 | ||
|
|
9eaa45f922 | ||
|
|
3d5e0c5832 | ||
|
|
71a36763df | ||
|
|
07bd3137ef | ||
|
|
c2b40b4dd3 | ||
|
|
4b16718602 | ||
|
|
2bf7e09862 | ||
|
|
db74e46fc3 | ||
|
|
13bba59ba6 | ||
|
|
2f5ebf1dd1 | ||
|
|
8f157e45bb | ||
|
|
6b259e2731 | ||
|
|
200660aba5 | ||
|
|
065c12cfbb | ||
|
|
318972620f | ||
|
|
e7f54d1592 | ||
|
|
a8571282b2 | ||
|
|
fc28062052 | ||
|
|
cc6306aa32 | ||
|
|
f0caa81253 | ||
|
|
cd98871f8f | ||
|
|
66e0d46d89 | ||
|
|
1dd5642e46 | ||
|
|
9ca34ca306 | ||
|
|
69f39a0bc3 | ||
|
|
50cf74db0a | ||
|
|
6886d8ff65 | ||
|
|
c1d118c1d4 | ||
|
|
5e46d63c42 | ||
|
|
863a59a8e2 | ||
|
|
c37fcf136a | ||
|
|
02a2292115 | ||
|
|
a483bc9837 | ||
|
|
120a6b85f4 | ||
|
|
9fea774a93 | ||
|
|
bf1e619ead | ||
|
|
0f8fcfc43e | ||
|
|
698b7fda06 | ||
|
|
23caa6e20f | ||
|
|
34e39c996e | ||
|
|
16ed20cfae | ||
|
|
ef368ceafa | ||
|
|
45480ef32c | ||
|
|
4de69029e5 | ||
|
|
c065770f48 | ||
|
|
27957ea051 | ||
|
|
94e9d1ab3b | ||
|
|
a374a9af35 | ||
|
|
3e80416eb6 | ||
|
|
b35c6c6d22 | ||
|
|
69d26cfee6 | ||
|
|
e3be1540f1 | ||
|
|
8e2b0d10e5 | ||
|
|
57c5761920 | ||
|
|
0841ff5feb | ||
|
|
45115384b5 | ||
|
|
dae2563850 | ||
|
|
fb4df5a0b7 | ||
|
|
b2f0303d1e | ||
|
|
f8b2a0b4d8 | ||
|
|
e1fcb78ce3 | ||
|
|
d6309088c0 | ||
|
|
1fb3a2e28f | ||
|
|
df25d4e03e | ||
|
|
9c2f0287e0 | ||
|
|
9912d41714 | ||
|
|
8b2cd87d9e | ||
|
|
3e6d23ae7e | ||
|
|
c96c39d5b1 | ||
|
|
a4556e90cd | ||
|
|
4d2c4b4423 | ||
|
|
530de3f031 | ||
|
|
c9622f6fd4 | ||
|
|
854628d959 | ||
|
|
35dbf6c44b | ||
|
|
b9fec7436f | ||
|
|
808d19c374 | ||
|
|
441d7205ed | ||
|
|
8b3d3b68c6 | ||
|
|
aa0e038ef7 | ||
|
|
5e5e5a35d9 | ||
|
|
10e35a55ea | ||
|
|
83cebea780 | ||
|
|
91bbb92c50 | ||
|
|
c7dd6ff28a | ||
|
|
df761a99ce | ||
|
|
359416e2b6 | ||
|
|
e140c0d60c | ||
|
|
0030971f6f | ||
|
|
3a90aaf1e6 | ||
|
|
f8a199af49 | ||
|
|
3a5de8e10c | ||
|
|
3c881809f4 | ||
|
|
8b6e9e08c0 | ||
|
|
3c8da3e3b4 | ||
|
|
0a39d909b2 | ||
|
|
d35d1092a4 | ||
|
|
7ae655c56a | ||
|
|
58f35ba413 | ||
|
|
e61eb24ec2 | ||
|
|
2e93d2ce51 | ||
|
|
9d21e1efd5 | ||
|
|
e0e6b3ad6b | ||
|
|
815cdc5b3c | ||
|
|
bc20f1e684 | ||
|
|
20e5f2bec6 | ||
|
|
c3b6fa55b6 | ||
|
|
69045db3a9 | ||
|
|
7b2240c80b | ||
|
|
dcfbd90dd7 | ||
|
|
70e6ab5782 | ||
|
|
175879823f | ||
|
|
2271a90adb | ||
|
|
2bdde5845e | ||
|
|
0c40497a01 | ||
|
|
45dff0f550 | ||
|
|
e9035ef6ee | ||
|
|
02ca94e6e6 | ||
|
|
432b7d2dc8 | ||
|
|
181d315d2c | ||
|
|
d5f7e616eb | ||
|
|
9a8869e8a4 | ||
|
|
85d56ed76f | ||
|
|
06c827b5c8 | ||
|
|
41259ff361 | ||
|
|
cccee1a668 | ||
|
|
c46b35362b | ||
|
|
2f96a6d8bf | ||
|
|
c78a47a3e5 | ||
|
|
b049721683 | ||
|
|
11c06fe9fe | ||
|
|
c5f6e53d0d | ||
|
|
1184672bb1 | ||
|
|
57d3a2ba35 | ||
|
|
1d813b0183 | ||
|
|
9b24931518 | ||
|
|
7e5b8b7bdf | ||
|
|
7e36473aff | ||
|
|
adbb512306 | ||
|
|
b951f4ad4b | ||
|
|
48900662ae | ||
|
|
da31a66c07 | ||
|
|
ca710b1cbb | ||
|
|
a4d7eec145 | ||
|
|
232c2fe87f | ||
|
|
661112cfd4 | ||
|
|
d4a84eaa9a | ||
|
|
99f8af64d0 | ||
|
|
1541ea9ffc | ||
|
|
ced86e693c | ||
|
|
53920d5bd3 | ||
|
|
15c5a9dac0 | ||
|
|
d63cfdbe7d | ||
|
|
569461a01c | ||
|
|
37c7dee236 | ||
|
|
cc230091c8 | ||
|
|
bac33cf246 | ||
|
|
7a9c0506b4 | ||
|
|
fbd7c15a4b | ||
|
|
64edf24bc7 | ||
|
|
3f456d683f | ||
|
|
aac0824fd3 | ||
|
|
1b32ca0b93 | ||
|
|
c29456aac9 | ||
|
|
715f25d059 | ||
|
|
c85e31ec7a | ||
|
|
0d6bfc4fa4 | ||
|
|
106917add2 | ||
|
|
1e10a2bac3 | ||
|
|
63dd09cd6f | ||
|
|
c7e6935f42 | ||
|
|
1be5054e86 | ||
|
|
f40755aca7 | ||
|
|
d49b78cf34 | ||
|
|
10e80ae064 | ||
|
|
f3ebb214aa | ||
|
|
dc41c3bd9f | ||
|
|
56ff09f3ac | ||
|
|
1707b27d14 | ||
|
|
7c92963eca | ||
|
|
ecf82c90ee | ||
|
|
580f06fe00 | ||
|
|
5a403b7765 | ||
|
|
f7367e56af | ||
|
|
f066abc151 | ||
|
|
2bee92f332 | ||
|
|
7d9ed4e1bf | ||
|
|
24696e6b98 | ||
|
|
71f658b07d | ||
|
|
920f56353f | ||
|
|
e0fe9167ea | ||
|
|
45a2349a0d | ||
|
|
d7b0e8469e | ||
|
|
8afc3b8e23 | ||
|
|
9caa63d5b2 | ||
|
|
1d32df91ce | ||
|
|
fa973f65bf | ||
|
|
c027f02e5a | ||
|
|
9cee0126d7 | ||
|
|
c713602d56 | ||
|
|
c8293cbbda | ||
|
|
a9c51388cf | ||
|
|
ad1d65e91a | ||
|
|
3a6c7803e8 | ||
|
|
aa837eddd5 | ||
|
|
faef57838f | ||
|
|
04d4c5e017 | ||
|
|
9158877569 | ||
|
|
1eae07f1b8 | ||
|
|
14b22487f1 | ||
|
|
1ac7cd5835 | ||
|
|
5336f01725 | ||
|
|
b8b66b1829 | ||
|
|
fd3e988a20 | ||
|
|
b1d98f4e58 | ||
|
|
9e7daf61d0 | ||
|
|
6fbe25753b | ||
|
|
03f0edc5b5 | ||
|
|
5536f1e835 | ||
|
|
0de36706da | ||
|
|
17722dad6d | ||
|
|
0371599996 | ||
|
|
199649b30f | ||
|
|
4ef35488db | ||
|
|
70a91ee6ce | ||
|
|
418a27e47e | ||
|
|
61c76d02cc | ||
|
|
d98641221d | ||
|
|
8a14f87a44 | ||
|
|
02ce71734c | ||
|
|
96c2743280 | ||
|
|
7bfc34b51c | ||
|
|
40d820fd05 | ||
|
|
d69287aaf7 | ||
|
|
d475b0ba9e | ||
|
|
1638b744b7 | ||
|
|
8b19894a06 | ||
|
|
d2e0dc99de | ||
|
|
d04e40b5fd | ||
|
|
75d797b5cd | ||
|
|
ecf4891087 | ||
|
|
0e1a418678 | ||
|
|
5bef13df94 | ||
|
|
d8386121a8 | ||
|
|
000677abb6 | ||
|
|
64eb87e9b5 | ||
|
|
aa5e92bee2 | ||
|
|
0bf79dc5d6 | ||
|
|
adb2171c0a | ||
|
|
d6f8923f86 | ||
|
|
cf91ab9d5f | ||
|
|
a0fb9531db | ||
|
|
eca9353b28 | ||
|
|
a259730639 | ||
|
|
2e93d10eba | ||
|
|
70a3ceb64e | ||
|
|
b726f60afd | ||
|
|
2fa1a64999 | ||
|
|
a42b659af9 | ||
|
|
004c3230a4 | ||
|
|
2332c41510 | ||
|
|
ec3039c5a2 | ||
|
|
639d6e6071 | ||
|
|
cd518d4726 | ||
|
|
b7d9c00dff | ||
|
|
4b17575f5a | ||
|
|
5ba4bba138 | ||
|
|
b3ee5dba0f | ||
|
|
d87ff5afa9 | ||
|
|
62a24bd38f | ||
|
|
8d373c15b8 | ||
|
|
671f3e74a4 | ||
|
|
7e810233d9 | ||
|
|
4700dbd676 | ||
|
|
74e18f4317 | ||
|
|
1eea95cf18 | ||
|
|
7291b10727 | ||
|
|
6804916697 | ||
|
|
819e61bf14 | ||
|
|
b8f7e4c8ec | ||
|
|
17bcc0eed4 | ||
|
|
13003da289 | ||
|
|
9273538955 | ||
|
|
9750189def | ||
|
|
cb17ee9871 | ||
|
|
4c3b78ba9a | ||
|
|
e00b6a401b | ||
|
|
ac0ab8a7b4 | ||
|
|
0aff3941f4 | ||
|
|
27b022d4d9 | ||
|
|
e188928742 | ||
|
|
780e3c7fb7 | ||
|
|
1b5146d3ac | ||
|
|
80cf3ca6b9 | ||
|
|
2272b30a91 | ||
|
|
b5fb1cb07c | ||
|
|
4ea34a9c22 | ||
|
|
48e7de9f9e | ||
|
|
76dd2369a7 | ||
|
|
78e0cd6e77 | ||
|
|
5514a04cb4 | ||
|
|
99ca78b235 | ||
|
|
ab45db1665 | ||
|
|
bb38bcb67d | ||
|
|
6cc2912542 | ||
|
|
340b2ca624 | ||
|
|
5baa15de03 | ||
|
|
6ddca804d1 | ||
|
|
f9831a85fb | ||
|
|
7261033b7f | ||
|
|
3ad6866198 | ||
|
|
1c7d4165ab | ||
|
|
3e48b1a8ac | ||
|
|
07be100daf | ||
|
|
de9351eefb | ||
|
|
2c44b5b3a1 | ||
|
|
136f1e2fc7 | ||
|
|
ad39add55f | ||
|
|
f3c301e359 | ||
|
|
1c37a1b4d6 | ||
|
|
7222529904 | ||
|
|
f26eccd00f | ||
|
|
73375a76ac | ||
|
|
d4416d200e | ||
|
|
d1b235dd83 | ||
|
|
3ac5e0423a | ||
|
|
fc6de5f3c0 | ||
|
|
b1e475d81d | ||
|
|
4a09a4324f | ||
|
|
47f94327c5 | ||
|
|
a009ed0b6b | ||
|
|
2476a686e7 | ||
|
|
f0db93773f | ||
|
|
fabe824c8b | ||
|
|
78a077397e | ||
|
|
0c4b456aaa | ||
|
|
09185167bc | ||
|
|
5d78c3203c | ||
|
|
fa5322d3f9 | ||
|
|
d21aa5cac2 | ||
|
|
102d5c57cb | ||
|
|
ffb4de9fd9 | ||
|
|
6b3d8886e5 | ||
|
|
ddc10272a0 | ||
|
|
0b5ef00165 | ||
|
|
2a50416fc3 | ||
|
|
ce514d9f83 | ||
|
|
9b77e7fd13 | ||
|
|
abb44d3327 | ||
|
|
11eaf3d48a | ||
|
|
76c2cc2c3e | ||
|
|
0e6c8bd12e | ||
|
|
a9fb008317 | ||
|
|
e9f3a5b3e4 | ||
|
|
ebc45dff45 | ||
|
|
fc4a5ebfd3 | ||
|
|
1d7b688c55 | ||
|
|
f14a5ffbbf | ||
|
|
cada0d593c | ||
|
|
0436540791 | ||
|
|
a87ac86e18 | ||
|
|
1278b23150 | ||
|
|
02f5ea4b9d | ||
|
|
2e14e613d0 | ||
|
|
85c2889652 | ||
|
|
23dd056b60 | ||
|
|
71043e372a | ||
|
|
c412d073b9 | ||
|
|
8fb03ff1b9 | ||
|
|
c2b6aef6f4 | ||
|
|
24547318c6 | ||
|
|
b693112c80 | ||
|
|
48d1184066 | ||
|
|
8da9ebc2e0 | ||
|
|
5ba510474b | ||
|
|
51214d1be1 | ||
|
|
4d6e15d7af | ||
|
|
4721894427 | ||
|
|
d429865b6e | ||
|
|
ca5881a72c | ||
|
|
7151b9daff | ||
|
|
d9b5e28b22 | ||
|
|
aa7954a7d6 | ||
|
|
802c70d1ab | ||
|
|
8f905988e9 | ||
|
|
478c5595ad | ||
|
|
3977e1f29e | ||
|
|
2b1ef97354 | ||
|
|
eaddf7f1a5 | ||
|
|
3237de3085 | ||
|
|
df3d398d31 | ||
|
|
b44b3401b7 | ||
|
|
c28ca0fac9 | ||
|
|
235e2b6c2c | ||
|
|
d68b84bc27 | ||
|
|
edca528608 | ||
|
|
b75e8f2abf | ||
|
|
ec3158e4cd | ||
|
|
9c8c8041e0 | ||
|
|
e3adaacb51 | ||
|
|
c49e11484f | ||
|
|
7b4b9a80fa | ||
|
|
c7ad066987 | ||
|
|
f4d229f1ba | ||
|
|
25a8a00771 | ||
|
|
a67f7422b2 | ||
|
|
ed8150cfb6 | ||
|
|
462a163ba7 | ||
|
|
6374175a64 | ||
|
|
280b15ba2a | ||
|
|
2a0b488e99 | ||
|
|
3ef7c4ab51 | ||
|
|
e72d746036 | ||
|
|
0bb4091e34 | ||
|
|
6822fc595c | ||
|
|
a506a589dd | ||
|
|
684a5977dd | ||
|
|
ac3682e058 | ||
|
|
d5faf01f5a | ||
|
|
ecba1b6838 | ||
|
|
8d8b029285 | ||
|
|
bf6f855868 | ||
|
|
aa6a499329 | ||
|
|
0971650ef9 | ||
|
|
64c4fdccf7 | ||
|
|
d715ffbc8e | ||
|
|
aef801b5b5 | ||
|
|
6c9e29796b | ||
|
|
364bb3ac1e | ||
|
|
428ea68507 | ||
|
|
5cf59408a7 | ||
|
|
1596843015 | ||
|
|
af6582ff5b | ||
|
|
1799d4c675 | ||
|
|
808e1c0330 | ||
|
|
dacd96cab5 | ||
|
|
ca4d3bf64d | ||
|
|
ea38b043c1 | ||
|
|
a39746df2e | ||
|
|
2367a8e50b | ||
|
|
cb121d7f17 | ||
|
|
412793c21d | ||
|
|
ce2286c48b | ||
|
|
d5694d6de0 | ||
|
|
121218aa8a | ||
|
|
6bb53fa758 | ||
|
|
89aa0c5471 | ||
|
|
53fcbf6afa | ||
|
|
2f5643ae6b | ||
|
|
d162ac8b3d | ||
|
|
c9a704fbde | ||
|
|
9d5a822a3a | ||
|
|
50eba4066a | ||
|
|
6116ae5330 | ||
|
|
447226576f | ||
|
|
3f8b872f17 | ||
|
|
e573ddc2db | ||
|
|
fe9aa681f0 | ||
|
|
1b2f2c1559 | ||
|
|
84c75a86c3 | ||
|
|
eedbde6f15 | ||
|
|
4e441e5a08 | ||
|
|
3e287a36c2 | ||
|
|
eadc477695 | ||
|
|
59aa324678 | ||
|
|
219bce1467 | ||
|
|
46bde401bd | ||
|
|
01beac4956 | ||
|
|
fcd981e6b7 | ||
|
|
825833cfcc | ||
|
|
4a4c49bf68 | ||
|
|
763cea423a | ||
|
|
0fee355ff5 | ||
|
|
6f6f3c9dc5 | ||
|
|
25e5d88ab2 | ||
|
|
87013340bb | ||
|
|
47c075ccc9 | ||
|
|
b1a32d4ccf | ||
|
|
7af6a8dbdf | ||
|
|
383e99e4ef | ||
|
|
8c7cfc4d11 | ||
|
|
496ee730c8 | ||
|
|
71f7ff5101 | ||
|
|
1ea00f68a2 | ||
|
|
8df7c2d84f | ||
|
|
46557a7a1f | ||
|
|
d8c2a8271f | ||
|
|
cc4c705fc0 | ||
|
|
22f249fcf6 | ||
|
|
c262362a03 | ||
|
|
212df9aa7b | ||
|
|
691e39ec76 | ||
|
|
107cae2975 | ||
|
|
6742e0c376 | ||
|
|
8f70137b1a | ||
|
|
707db51b1b | ||
|
|
5b5fa1aa29 | ||
|
|
2b9cc9666a | ||
|
|
82eba22292 | ||
|
|
5c84e8f23c | ||
|
|
081b61677a | ||
|
|
9c54814b98 | ||
|
|
35d7b855ed | ||
|
|
0b8799274c | ||
|
|
ace2b737d8 | ||
|
|
ad85268524 | ||
|
|
832a320e22 | ||
|
|
83763df6fd | ||
|
|
bee868e9ba | ||
|
|
c41de81694 | ||
|
|
41aaeb1ff0 | ||
|
|
16be2792ab | ||
|
|
6610bb355c | ||
|
|
4312fd7291 | ||
|
|
89a225a96d | ||
|
|
a590977639 | ||
|
|
0d0d116bde | ||
|
|
341bdb5a54 | ||
|
|
2f3dbfb289 | ||
|
|
169cfbbeed | ||
|
|
f97a4afd8f | ||
|
|
868e4a6d81 | ||
|
|
6f48f7d3ac | ||
|
|
1ed3ecb409 | ||
|
|
2cb455b9d4 | ||
|
|
d4b5bf0f78 | ||
|
|
69013772c1 | ||
|
|
3448c83431 | ||
|
|
e4c84542ea | ||
|
|
acddc0323b | ||
|
|
c4285f0d30 | ||
|
|
977d6dd247 | ||
|
|
2bb27fffb7 | ||
|
|
0261ed353d | ||
|
|
96fecfd7c5 | ||
|
|
9fac1b8105 | ||
|
|
95fbcd7b9a | ||
|
|
6adf227611 | ||
|
|
b5cb429243 | ||
|
|
26ba8079a3 | ||
|
|
f999d30bc5 | ||
|
|
ecb1cc4ed4 | ||
|
|
5622bcae16 | ||
|
|
f4539ee289 | ||
|
|
d2138694b4 | ||
|
|
1767e21273 | ||
|
|
8f9d799342 | ||
|
|
a3b0b246f4 | ||
|
|
dee85f14fe | ||
|
|
27309114be | ||
|
|
704afed97b | ||
|
|
121f0a2c6c | ||
|
|
1c580ec92c | ||
|
|
ab8fc721a0 | ||
|
|
790447115c | ||
|
|
80abeac28a | ||
|
|
54915f87ce | ||
|
|
f34f1309a7 | ||
|
|
0ad52b7d19 | ||
|
|
809f60df06 | ||
|
|
cbdcd8253c | ||
|
|
7ac2cd7cc8 | ||
|
|
cf1bb1348c | ||
|
|
b60a26ff9e | ||
|
|
b805c07342 | ||
|
|
8d69f539ac | ||
|
|
c8c0054f67 | ||
|
|
1085385bbe | ||
|
|
56460b220c | ||
|
|
a583ebe590 | ||
|
|
ce8175a800 | ||
|
|
c987e1ef44 | ||
|
|
b36ec152d2 | ||
|
|
44c62e703e | ||
|
|
0f59c1d5e3 | ||
|
|
5739f0b459 | ||
|
|
4145fabfb6 |
No files matched your search
@@ -64,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -188,6 +188,40 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Install
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: Test GL No-Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
|
||||
|
||||
- name: No thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
|
||||
|
||||
- name: Test GL Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
|
||||
|
||||
- name: Thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
@@ -207,4 +241,3 @@ jobs:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
name: Vixl Simulator run
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
# Only the x86-64 runner is fast enough to run this
|
||||
arch: [[self-hosted, x64], [self-hosted, ARMv8.4]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"ThunksDB": {
|
||||
"GL": 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"ThunksDB": {
|
||||
"Vulkan": 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
if(NOT EXISTS "@CMAKE_BINARY_DIR@/install_manifest.txt")
|
||||
message(FATAL_ERROR "Cannot find install manifest: @CMAKE_BINARY_DIR@/install_manifest.txt")
|
||||
endif()
|
||||
|
||||
file(READ "@CMAKE_BINARY_DIR@/install_manifest.txt" files)
|
||||
string(REGEX REPLACE "\n" ";" files "${files}")
|
||||
foreach(file ${files})
|
||||
message(STATUS "Uninstalling $ENV{DESTDIR}${file}")
|
||||
if(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
|
||||
exec_program(
|
||||
"@CMAKE_COMMAND@" ARGS "-E remove \"$ENV{DESTDIR}${file}\""
|
||||
OUTPUT_VARIABLE rm_out
|
||||
RETURN_VALUE rm_retval
|
||||
)
|
||||
if(NOT "${rm_retval}" STREQUAL 0)
|
||||
message(FATAL_ERROR "Problem when removing $ENV{DESTDIR}${file}")
|
||||
endif()
|
||||
else(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
|
||||
message(STATUS "File $ENV{DESTDIR}${file} does not exist.")
|
||||
endif()
|
||||
endforeach()
|
||||
+73
-6
@@ -7,6 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -27,10 +28,37 @@ option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CMakeFiles/cmake_uninstall.cmake.in"
|
||||
"${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake"
|
||||
IMMEDIATE @ONLY)
|
||||
|
||||
add_custom_target(uninstall
|
||||
COMMAND ${CMAKE_COMMAND} -P ${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake)
|
||||
endif()
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
@@ -84,10 +112,9 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
@@ -363,8 +390,6 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
include_directories(FEXHeaderUtils/)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
@@ -403,8 +428,28 @@ if (BUILD_THUNKS)
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DBITNESS=64"
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs-32
|
||||
PREFIX guest-libs-32
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest_32"
|
||||
CMAKE_ARGS
|
||||
"-DBITNESS=32"
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
@@ -422,6 +467,28 @@ if (BUILD_THUNKS)
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs-32
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_guest-libs)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
|
||||
+74
-66
@@ -6,10 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -18,17 +18,17 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
@@ -37,8 +37,8 @@
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
@@ -48,121 +48,129 @@
|
||||
"xcb": {
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb_dri2-guest.so",
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb_dri3-guest.so",
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb_xfixes-guest.so",
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb_shm-guest.so",
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb_sync-guest.so",
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb_randr-guest.so",
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb_present-guest.so",
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb_glx-guest.so",
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libshmfence-guest.so",
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
Vendored
+14
-4
@@ -9,12 +9,19 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
endif()
|
||||
|
||||
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
|
||||
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# If the vixl simulator is enabled then we are using the ARM64 JIT
|
||||
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
|
||||
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
|
||||
else()
|
||||
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
|
||||
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
|
||||
endif()
|
||||
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
@@ -27,7 +34,6 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
|
||||
include(CheckCXXCompilerFlag)
|
||||
include(CheckIncludeFileCXX)
|
||||
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
# Useful to have for freestanding libFEXCore
|
||||
add_subdirectory(External/vixl/)
|
||||
@@ -76,3 +82,7 @@ add_subdirectory(Source/)
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
+14
-5
@@ -135,7 +135,6 @@ set (SRCS
|
||||
Interface/IR/Passes/PhiValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
@@ -143,6 +142,7 @@ set (SRCS
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
@@ -177,6 +177,15 @@ if (_M_ARM_64)
|
||||
list(APPEND DEFINES -D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# We can run the simulator on both x86-64 or AArch64 hosts
|
||||
list(APPEND DEFINES -DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER)
|
||||
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JIT_X86_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/JIT/x86_64/JIT.cpp
|
||||
@@ -213,7 +222,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
@@ -359,14 +368,14 @@ endfunction()
|
||||
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base fmt::fmt tiny-json)
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
@@ -374,7 +383,7 @@ endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
+10
-6
@@ -211,10 +211,12 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
|
||||
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
|
||||
@@ -629,7 +631,7 @@ namespace JSON {
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, bool Global);
|
||||
explicit AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
@@ -681,8 +683,10 @@ namespace JSON {
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, bool Global)
|
||||
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
|
||||
AppLoader::AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
@@ -754,8 +758,8 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
|
||||
+17
-2
@@ -16,10 +16,11 @@
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Default": "false",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation"
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
]
|
||||
},
|
||||
"MaxInst": {
|
||||
@@ -90,6 +91,20 @@
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkHostLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit host-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkGuestLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
|
||||
+3
-1
@@ -15,6 +15,7 @@
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
@@ -104,6 +105,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(DumpIR, DUMPIR);
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
@@ -189,7 +191,7 @@ namespace FEXCore::Context {
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -572,7 +571,7 @@ bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
PC[1] = STP;
|
||||
PC[2] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[0], 16);
|
||||
FEXCore::ARMEmitter::Buffer::ClearICache(&PC[0], 16);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -2311,7 +2310,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
return false;
|
||||
}
|
||||
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[-1], 16);
|
||||
FEXCore::ARMEmitter::Buffer::ClearICache(&PC[-1], 16);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
|
||||
+212
-154
@@ -1,4 +1,5 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
@@ -17,29 +18,27 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : nullptr, size)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
if (ctx->HostFeatures.SupportsAtomics) {
|
||||
// Hypervisor can hide this on the c630?
|
||||
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
|
||||
}
|
||||
|
||||
SetCPUFeatures(Features);
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = Reg.IsX();
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::munmap(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
if (Is64Bit && ((~Constant)>> 16) == 0) {
|
||||
movn(Reg, (~Constant) & 0xFFFF);
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
@@ -86,17 +85,17 @@ void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant,
|
||||
else {
|
||||
// Need to use ADRP + ADD
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
add(Reg, Reg, Constant & 0xFFF);
|
||||
add(s, Reg, Reg, Constant & 0xFFF);
|
||||
NumMoves = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
movz(s, Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
movk(s, Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
}
|
||||
}
|
||||
@@ -112,125 +111,143 @@ void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant,
|
||||
void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
MemOperand PairOffset(sp, -16, PreIndex);
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x19, x20},
|
||||
{x21, x22},
|
||||
{x23, x24},
|
||||
{x25, x26},
|
||||
{x27, x28},
|
||||
{x29, x30},
|
||||
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
|
||||
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
|
||||
{ARMEmitter::XReg::x25, ARMEmitter::XReg::x26},
|
||||
{ARMEmitter::XReg::x27, ARMEmitter::XReg::x28},
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
}};
|
||||
|
||||
for (auto &RegPair : CalleeSaved) {
|
||||
stp(RegPair.first, RegPair.second, PairOffset);
|
||||
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
|
||||
}
|
||||
|
||||
// Additionally we need to store the lower 64bits of v8-v15
|
||||
// Here's a fun thing, we can use two ST4 instructions to store everything
|
||||
// We just need a single sub to sp before that
|
||||
const std::array<
|
||||
std::tuple<vixl::aarch64::VRegister,
|
||||
vixl::aarch64::VRegister,
|
||||
vixl::aarch64::VRegister,
|
||||
vixl::aarch64::VRegister>, 2> FPRs = {{
|
||||
{v8, v9, v10, v11},
|
||||
{v12, v13, v14, v15},
|
||||
std::tuple<ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
}};
|
||||
|
||||
uint32_t VectorSaveSize = sizeof(uint64_t) * 8;
|
||||
sub(sp, sp, VectorSaveSize);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, VectorSaveSize);
|
||||
// SP supporting move
|
||||
// We just saved x19 so it is safe
|
||||
add(x19, sp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
MemOperand QuadOffset(x19, 32, PostIndex);
|
||||
for (auto &RegQuad : FPRs) {
|
||||
st4(std::get<0>(RegQuad).D(),
|
||||
std::get<1>(RegQuad).D(),
|
||||
std::get<2>(RegQuad).D(),
|
||||
std::get<3>(RegQuad).D(),
|
||||
st4(ARMEmitter::SubRegSize::i64Bit,
|
||||
std::get<0>(RegQuad),
|
||||
std::get<1>(RegQuad),
|
||||
std::get<2>(RegQuad),
|
||||
std::get<3>(RegQuad),
|
||||
0,
|
||||
QuadOffset);
|
||||
ARMEmitter::Reg::r19,
|
||||
32);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
const std::array<
|
||||
std::tuple<vixl::aarch64::VRegister,
|
||||
vixl::aarch64::VRegister,
|
||||
vixl::aarch64::VRegister,
|
||||
vixl::aarch64::VRegister>, 2> FPRs = {{
|
||||
{v12, v13, v14, v15},
|
||||
{v8, v9, v10, v11},
|
||||
std::tuple<ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
|
||||
}};
|
||||
|
||||
MemOperand QuadOffset(sp, 32, PostIndex);
|
||||
for (auto &RegQuad : FPRs) {
|
||||
ld4(std::get<0>(RegQuad).D(),
|
||||
std::get<1>(RegQuad).D(),
|
||||
std::get<2>(RegQuad).D(),
|
||||
std::get<3>(RegQuad).D(),
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit,
|
||||
std::get<0>(RegQuad),
|
||||
std::get<1>(RegQuad),
|
||||
std::get<2>(RegQuad),
|
||||
std::get<3>(RegQuad),
|
||||
0,
|
||||
QuadOffset);
|
||||
ARMEmitter::Reg::rsp,
|
||||
32);
|
||||
}
|
||||
|
||||
MemOperand PairOffset(sp, 16, PostIndex);
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x29, x30},
|
||||
{x27, x28},
|
||||
{x25, x26},
|
||||
{x23, x24},
|
||||
{x21, x22},
|
||||
{x19, x20},
|
||||
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
{ARMEmitter::XReg::x27, ARMEmitter::XReg::x28},
|
||||
{ARMEmitter::XReg::x25, ARMEmitter::XReg::x26},
|
||||
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
|
||||
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
}};
|
||||
|
||||
for (auto &RegPair : CalleeSaved) {
|
||||
ldp(RegPair.first, RegPair.second, PairOffset);
|
||||
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & GPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & GPRSpillMask)) {
|
||||
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & GPRSpillMask)) {
|
||||
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & GPRSpillMask)) {
|
||||
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
|
||||
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
}
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Reg, PRED_TMP_32B, STATE.R(), TMP4.R());
|
||||
}
|
||||
} else {
|
||||
}
|
||||
} else {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
auto TmpReg = SRA64[__builtin_ffs(GPRSpillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -239,96 +256,137 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & GPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & GPRFillMask)) {
|
||||
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & GPRFillMask)) {
|
||||
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & GPRFillMask)) {
|
||||
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg, PRED_TMP_32B, STATE.R(), TMP4.R());
|
||||
}
|
||||
} else {
|
||||
}
|
||||
} else {
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = SRA64[__builtin_ffs(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
else if (((1U << Reg1.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
else if (((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if ((1U << Reg1.Idx()) & GPRFillMask) {
|
||||
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if ((1U << Reg2.Idx()) & GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR() {
|
||||
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = 1 * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(sp, sp, SPOffset);
|
||||
int i = 0;
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
for (auto RA : RAFPR)
|
||||
{
|
||||
str(RA.Q(), MemOperand(sp, i * 8));
|
||||
i+=2;
|
||||
// rsp capable move
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
st4b(Reg1, Reg2, Reg3, Reg4, PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
static_assert(RAFPR.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
|
||||
#if 0 // All GPRs should be caller saved
|
||||
for (auto RA : RA64)
|
||||
{
|
||||
str(RA, MemOperand(sp, i * 8));
|
||||
i++;
|
||||
}
|
||||
#endif
|
||||
|
||||
str(lr, MemOperand(sp, i * 8));
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
|
||||
int i = 0;
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
for (auto RA : RAFPR)
|
||||
{
|
||||
ldr(RA.Q(), MemOperand(sp, i * 8));
|
||||
i+=2;
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
ld4b(Reg1, Reg2, Reg3, Reg4, PRED_TMP_32B, ARMEmitter::Reg::rsp);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
|
||||
}
|
||||
}
|
||||
|
||||
#if 0 // All GPRs should be caller saved
|
||||
for (auto RA : RA64)
|
||||
{
|
||||
ldr(RA, MemOperand(sp, i * 8));
|
||||
i++;
|
||||
}
|
||||
#endif
|
||||
|
||||
ldr(lr, MemOperand(sp, i * 8));
|
||||
|
||||
add(sp, sp, SPOffset);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
|
||||
+128
-30
@@ -1,5 +1,9 @@
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
@@ -8,6 +12,13 @@
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#endif
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
@@ -17,56 +28,74 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
// All but x29 are caller saved
|
||||
const std::array<aarch64::Register, 16> SRA64 = {
|
||||
x4, x5, x6, x7, x8, x9, x10, x11,
|
||||
x12, x18, x17, x16, x15, x14, x13, x29
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9, FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r18, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r15, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r13, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
// All are callee saved
|
||||
const std::array<aarch64::Register, 9> RA64 = {
|
||||
x20, x21, x22, x23, x24, x25, x26, x27,
|
||||
x19
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r19
|
||||
};
|
||||
|
||||
const std::array<std::pair<aarch64::Register, aarch64::Register>, 4> RA64Pair = {{
|
||||
{x20, x21},
|
||||
{x22, x23},
|
||||
{x24, x25},
|
||||
{x26, x27},
|
||||
}};
|
||||
|
||||
const std::array<std::pair<aarch64::Register, aarch64::Register>, 4> RA32Pair = {{
|
||||
{w20, w21},
|
||||
{w22, w23},
|
||||
{w24, w25},
|
||||
{w26, w27},
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
const std::array<aarch64::VRegister, 16> SRAFPR = {
|
||||
v16, v17, v18, v19, v20, v21, v22, v23,
|
||||
v24, v25, v26, v27, v28, v29, v30, v31
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
/*v0, v1, v2, v3,*/v4, v5, v6, v7, // v0 ~ v3 are used as temps
|
||||
v8, v9, v10, v11, v12, v13, v14, v15
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
/*FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,*/FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, // FEXCore::ARMEmitter::VReg::v0 ~ FEXCore::ARMEmitter::VReg::v3 are used as temps
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
// GPR temporaries. Only x3 can be used across spill boundaries
|
||||
// so if these ever need to change, be very careful about that.
|
||||
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
|
||||
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
|
||||
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
|
||||
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
|
||||
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
|
||||
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
|
||||
constexpr auto VTMP3 = FEXCore::ARMEmitter::VReg::v2;
|
||||
constexpr auto VTMP4 = FEXCore::ARMEmitter::VReg::v3;
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
void LoadConstant(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
@@ -76,7 +105,7 @@ protected:
|
||||
// We can't guarantee only the lower 64bits are used so flush everything
|
||||
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
|
||||
|
||||
void PushDynamicRegsAndLR();
|
||||
void PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegsAndLR();
|
||||
|
||||
void PushCalleeSavedRegisters();
|
||||
@@ -84,6 +113,75 @@ protected:
|
||||
|
||||
void Align16B();
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Generates a vixl simulator runtime call.
|
||||
//
|
||||
// This matches behaviour of vixl's macro assembler, but we need to reimplement it since we aren't using the macro assembler.
|
||||
// This isn't too complex with how vixl emits this.
|
||||
//
|
||||
// Emit:
|
||||
// 1) hlt(kRuntimeCallOpcode)
|
||||
// 2) Simulator wrapper handler
|
||||
// 3) Function to call
|
||||
// 4) Style of the function call (Call versus tail-call)
|
||||
|
||||
template<typename R, typename... P>
|
||||
void GenerateRuntimeCall(R (*Function)(P...)) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
|
||||
|
||||
hlt(vixl::aarch64::kRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc64(SimulatorWrapperAddress);
|
||||
|
||||
// Runtime function address to call
|
||||
dc64(FunctionAddress);
|
||||
|
||||
// Call type
|
||||
dc32(vixl::aarch64::kCallRuntime);
|
||||
}
|
||||
|
||||
template<typename R, typename... P>
|
||||
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc64(SimulatorWrapperAddress);
|
||||
|
||||
// Register that contains the function to call
|
||||
dc32(Reg.Idx());
|
||||
|
||||
// Call type
|
||||
dc32(vixl::aarch64::kCallRuntime);
|
||||
}
|
||||
|
||||
template<>
|
||||
void GenerateIndirectRuntimeCall<float, __uint128_t>(ARMEmitter::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
|
||||
|
||||
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc64(SimulatorWrapperAddress);
|
||||
|
||||
// Register that contains the function to call
|
||||
dc32(Reg.Idx());
|
||||
|
||||
// Call type
|
||||
dc32(vixl::aarch64::kCallRuntime);
|
||||
}
|
||||
|
||||
#endif
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
vixl::aarch64::PrintDisassembler Disasm {stderr};
|
||||
#endif
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,912 @@
|
||||
/* ALU instruction emitters.
|
||||
*
|
||||
* Almost all of these operations have `ARMEmitter::Size` as their first argument.
|
||||
* This allows both 32-bit and 64-bit selection of how that instruction is going to operate.
|
||||
*
|
||||
* Some emitter operations explicitly use `XRegister` or `WRegister`.
|
||||
* This is usually due to the instruction only supporting one operating size.
|
||||
* Although in some cases is a minor convenience without any performance implications.
|
||||
*
|
||||
* FEX-Emu ALU operations usually have a 32-bit or 64-bit operating size encoded in the IR operation,
|
||||
* This allows FEX to use a single helper function which decodes to both handlers.
|
||||
*/
|
||||
public:
|
||||
// PC relative
|
||||
void adr(FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adr(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575, "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
void adr(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADR });
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
|
||||
void adr(FEXCore::ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adr(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
adr(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -4294967296 && Imm <= 4294963200 && (Imm & 0xFFF) == 0, "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADRP });
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adrp(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
adrp(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Add/subtract immediate
|
||||
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0001'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void adds(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0011'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0101'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void cmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0111'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, FEXCore::ARMEmitter::Reg::rsp, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void subs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0111'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
and_(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void bic(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
and_(s, rd, rn, ~Imm);
|
||||
}
|
||||
|
||||
void ands(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
ands(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void bics(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
ands(s, rd, rn, ~Imm);
|
||||
}
|
||||
|
||||
void orr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
orr(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void eor(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
// Move wide immediate
|
||||
void movn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
|
||||
LOGMAN_THROW_A_FMT((Offset % 16) == 0, "Offset must be 16bit aligned");
|
||||
|
||||
constexpr uint32_t Op = 0b001'0010'100 << 21;
|
||||
DataProcessing_MoveWide(Op, s, rd, Imm, Offset >> 4);
|
||||
}
|
||||
void mov(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
|
||||
movz(s, rd, Imm, 0);
|
||||
}
|
||||
void mov(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm) {
|
||||
movz(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, 0);
|
||||
}
|
||||
void mov(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm) {
|
||||
movz(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, 0);
|
||||
}
|
||||
|
||||
void movz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
|
||||
LOGMAN_THROW_A_FMT((Offset % 16) == 0, "Offset must be 16bit aligned");
|
||||
|
||||
constexpr uint32_t Op = 0b101'0010'100 << 21;
|
||||
DataProcessing_MoveWide(Op, s, rd, Imm, Offset >> 4);
|
||||
}
|
||||
void movk(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
|
||||
LOGMAN_THROW_A_FMT((Offset % 16) == 0, "Offset must be 16bit aligned");
|
||||
|
||||
constexpr uint32_t Op = 0b111'0010'100 << 21;
|
||||
DataProcessing_MoveWide(Op, s, rd, Imm, Offset >> 4);
|
||||
}
|
||||
|
||||
void movn(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
movn(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, Offset);
|
||||
}
|
||||
void movz(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
movz(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, Offset);
|
||||
}
|
||||
void movk(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
movk(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, Offset);
|
||||
}
|
||||
void movn(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
movn(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, Offset);
|
||||
}
|
||||
void movz(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
movz(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, Offset);
|
||||
}
|
||||
void movk(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
movk(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, Offset);
|
||||
}
|
||||
|
||||
// Bitfield
|
||||
void sxtb(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
sbfm(s, rd, rn, 0, 7);
|
||||
}
|
||||
void sxth(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
sbfm(s, rd, rn, 0, 15);
|
||||
}
|
||||
void sxtw(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
|
||||
sbfm(ARMEmitter::Size::i64Bit, rd, rn, 0, 31);
|
||||
}
|
||||
void sbfx(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
LOGMAN_THROW_A_FMT(width > 0, "sbfx needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to sbfx a region larger than the register");
|
||||
sbfm(s, rd, rn, lsb, lsb + width - 1);
|
||||
}
|
||||
void asr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
}
|
||||
|
||||
void uxtb(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
ubfm(s, rd, rn, 0, 7);
|
||||
}
|
||||
void uxth(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
ubfm(s, rd, rn, 0, 15);
|
||||
}
|
||||
void uxtw(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
ubfm(s, rd, rn, 0, 31);
|
||||
}
|
||||
|
||||
void ubfm(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b0101'0011'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
|
||||
}
|
||||
|
||||
void lsl(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to asr a region larger than the register");
|
||||
ubfm(s, rd, rn, (RegSize - shift) % RegSize, RegSize - shift - 1);
|
||||
}
|
||||
void lsr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to asr a region larger than the register");
|
||||
ubfm(s, rd, rn, shift, RegSize - 1);
|
||||
}
|
||||
void ubfx(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
LOGMAN_THROW_A_FMT(width > 0, "ubfx needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to ubfx a region larger than the register");
|
||||
ubfm(s, rd, rn, lsb, lsb + width - 1);
|
||||
}
|
||||
|
||||
void bfi(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(width > 0, "sbfx needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to sbfx a region larger than the register");
|
||||
bfm(s, rd, rn, (RegSize - lsb) & (RegSize - 1), width - 1);
|
||||
}
|
||||
|
||||
// Extract
|
||||
void extr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b001'0011'100 << 21;
|
||||
LOGMAN_THROW_A_FMT(Imm < RegSizeInBits(s), "Tried to extr a region larger than the register");
|
||||
DataProcessing_Extract(Op, s, rd, rn, rm, Imm);
|
||||
}
|
||||
|
||||
void ror(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm < RegSizeInBits(s), "Tried to extr a region larger than the register");
|
||||
extr(s, rd, rn, rn, Imm);
|
||||
}
|
||||
|
||||
// Data processing - 2 source
|
||||
void udiv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void sdiv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
|
||||
void lslv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void lsrv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void asrv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void rorv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void crc32b(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32h(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32w(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cb(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32ch(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cw(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void irg(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0001'00U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void gmi(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0001'01U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void pacga(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0011'00U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32x(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'11U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cx(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'11U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void subps(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b011'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Data processing - 1 source
|
||||
void rbit(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev16(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'01U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10);
|
||||
DataProcessing_1Source(Op, FEXCore::ARMEmitter::Size::i32Bit, rd, rn);
|
||||
}
|
||||
void rev32(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10);
|
||||
DataProcessing_1Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn);
|
||||
}
|
||||
void clz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cls(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'01U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'11U << 10);
|
||||
DataProcessing_1Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn);
|
||||
}
|
||||
void rev(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10) |
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
|
||||
// TODO: PAUTH
|
||||
|
||||
// Logical - shifted register
|
||||
void mov(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
orr(s, rd, FEXCore::ARMEmitter::Reg::zr, rn, ARMEmitter::ShiftType::LSL, 0);
|
||||
}
|
||||
void mov(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
|
||||
orr(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), FEXCore::ARMEmitter::Reg::zr, rn.R(), ARMEmitter::ShiftType::LSL, 0);
|
||||
}
|
||||
void mov(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn) {
|
||||
orr(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), FEXCore::ARMEmitter::Reg::zr, rn.R(), ARMEmitter::ShiftType::LSL, 0);
|
||||
}
|
||||
|
||||
void mvn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
orn(s, rd, FEXCore::ARMEmitter::Reg::zr, rn, Shift, amt);
|
||||
}
|
||||
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b000'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void ands(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b110'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void bic(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b000'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void bics(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b110'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void orr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void orn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void eor(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b100'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void eon(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b100'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
// AddSub - shifted register
|
||||
void add(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
add(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void adds(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void neg(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(rd, FEXCore::ARMEmitter::XReg::zr, rm, Shift, amt);
|
||||
}
|
||||
void cmp(FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i64Bit, FEXCore::ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void subs(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void negs(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(rd, FEXCore::ARMEmitter::XReg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void add(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
add(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void adds(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void neg(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(rd, FEXCore::ARMEmitter::WReg::zr, rm, Shift, amt);
|
||||
}
|
||||
void cmp(FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i32Bit, FEXCore::ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void subs(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void negs(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(rd, FEXCore::ARMEmitter::WReg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b000'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void adds(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b010'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b100'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void neg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(s, rd, FEXCore::ARMEmitter::Reg::zr, rm, Shift, amt);
|
||||
}
|
||||
void cmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void subs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b110'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void negs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(s, rd, FEXCore::ARMEmitter::Reg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
// AddSub - extended register
|
||||
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift <= 4, "Shift amount is too large");
|
||||
constexpr uint32_t Op = 0b000'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void adds(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b010'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b100'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void subs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b110'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void cmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b110'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Option, Shift);
|
||||
}
|
||||
|
||||
// AddSub - with carry
|
||||
void adc(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = 0b0001'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
void adcs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = 0b0011'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
void sbc(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = 0b0101'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
void sbcs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = 0b0111'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
// Rotate right into flags
|
||||
// TODO
|
||||
// Evaluate into flags
|
||||
// TODO
|
||||
// Conditional compare - register
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
ConditionalCompare(Op, 0, 0b00, 0, s, rn, rm, flags, Cond);
|
||||
}
|
||||
void ccmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
ConditionalCompare(Op, 1, 0b00, 0, s, rn, rm, flags, Cond);
|
||||
}
|
||||
|
||||
// Conditional compare - immediate
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
LOGMAN_THROW_A_FMT((rm & ~0b1'1111) == 0, "Comparison imm too large");
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
ConditionalCompare(Op, 0, 0b10, 0, s, rn, rm, flags, Cond);
|
||||
}
|
||||
void ccmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
LOGMAN_THROW_A_FMT((rm & ~0b1'1111) == 0, "Comparison imm too large");
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
ConditionalCompare(Op, 1, 0b10, 0, s, rn, rm, flags, Cond);
|
||||
}
|
||||
|
||||
// Conditional select
|
||||
void csel(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 0, 0b00, s, rd, rn, rm, Cond);
|
||||
}
|
||||
void cset(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 0, 0b01, s, rd, FEXCore::ARMEmitter::Reg::zr, FEXCore::ARMEmitter::Reg::zr, static_cast<FEXCore::ARMEmitter::Condition>(FEXCore::ToUnderlying(Cond) ^ FEXCore::ToUnderlying(FEXCore::ARMEmitter::Condition::CC_NE)));
|
||||
}
|
||||
void csinc(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 0, 0b01, s, rd, rn, rm, Cond);
|
||||
}
|
||||
void csinv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 1, 0b00, s, rd, rn, rm, Cond);
|
||||
}
|
||||
void csneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 1, 0b01, s, rd, rn, rm, Cond);
|
||||
}
|
||||
|
||||
// Data processing - 3 source
|
||||
void madd(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'000U << 21;
|
||||
DataProcessing_3Source(Op, 0, s, rd, rn, rm, ra);
|
||||
}
|
||||
void mul(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
madd(s, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void msub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'000U << 21;
|
||||
DataProcessing_3Source(Op, 1, s, rd, rn, rm, ra);
|
||||
}
|
||||
void mneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
msub(s, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void smaddl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'001U << 21;
|
||||
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void smull(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
smaddl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void smsubl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'001U << 21;
|
||||
DataProcessing_3Source(Op, 1, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void smnegl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
smsubl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void smulh(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = 0b001'1011'010U << 21;
|
||||
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void umaddl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'101U << 21;
|
||||
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void umull(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
umaddl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void umsubl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'101U << 21;
|
||||
DataProcessing_3Source(Op, 1, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void umnegl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
umsubl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void umulh(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = 0b001'1011'110U << 21;
|
||||
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
|
||||
private:
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
}
|
||||
void ands(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b111'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
}
|
||||
void orr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b011'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
}
|
||||
void eor(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b101'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void sbfm(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b0001'0011'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
|
||||
}
|
||||
void bfm(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b0011'0011'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
|
||||
}
|
||||
// 4.1.64 - Data processing - Immediate
|
||||
void DataProcessing_PCRel_Imm(uint32_t Op, FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
|
||||
// Ensure the immediate is masked.
|
||||
Imm &= 0b1'1111'1111'1111'1111'1111U;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= (Imm & 0b11) << 29;
|
||||
Instr |= (Imm >> 2) << 5;
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void DataProcessing_AddSub_Imm(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12) {
|
||||
bool TooLarge = (Imm & ~0b1111'1111'1111U) != 0;
|
||||
if (TooLarge && !LSL12 && ((Imm >> 12) & ~0b1111'1111'1111U) == 0) {
|
||||
// We can convert an immediate
|
||||
TooLarge = false;
|
||||
LSL12 = true;
|
||||
Imm >>= 12;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(TooLarge == false, "Imm amount too large: 0x{:x}", Imm);
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= LSL12 << 22;
|
||||
Instr |= Imm << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Move Wide
|
||||
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= Imm << 5;
|
||||
Instr |= Offset << 21;
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void DataProcessing_Logical_Imm(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= n << 22;
|
||||
Instr |= immr << 16;
|
||||
Instr |= imms << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void DataProcessing_Extract(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
// Current ARMv8 spec hardcodes SF == N for this class of instructions.
|
||||
// Anythign else is undefined behaviour.
|
||||
const uint32_t N = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 22) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= N;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= Imm << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Data-processing - 2 source
|
||||
void DataProcessing_2Source(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Data processing - 1 source
|
||||
template<typename T>
|
||||
void DataProcessing_1Source(uint32_t Op, FEXCore::ARMEmitter::Size s, T rd, T rn) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= FEXCore::ToUnderlying(Shift) << 22;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= static_cast<uint32_t>(amt) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// AddSub - extended register
|
||||
void DataProcessing_Extended_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= FEXCore::ToUnderlying(Option) << 13;
|
||||
Instr |= static_cast<uint32_t>(Shift) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
// Conditional compare - register
|
||||
template<typename T>
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, uint32_t o3, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, T rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= o1 << 30;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= FEXCore::ToUnderlying(Cond) << 12;
|
||||
Instr |= o2 << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= o3 << 4;
|
||||
Instr |= FEXCore::ToUnderlying(flags);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, T rm, FEXCore::ARMEmitter::Condition Cond) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= o1 << 30;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= FEXCore::ToUnderlying(Cond) << 12;
|
||||
Instr |= o2 << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Data-processing - 3 source
|
||||
void DataProcessing_3Source(uint32_t Op, uint32_t Op0, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= Encode_rm(rm);
|
||||
Instr |= Op0 << 15;
|
||||
Instr |= Encode_ra(ra);
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
|
||||
+4023
File diff suppressed because it is too large.
Load diff
+322
@@ -0,0 +1,322 @@
|
||||
/* Branch instruction emitters.
|
||||
*
|
||||
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
|
||||
*/
|
||||
public:
|
||||
// Branches, Exception Generating and System instructions
|
||||
public:
|
||||
// Conditional branch immediate
|
||||
///< Branch conditional
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
}
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
b(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
///< Branch consistent conditional
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
}
|
||||
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
bc(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void br(FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'000 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void blr(FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'001 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void ret(FEXCore::ARMEmitter::Register rn = FEXCore::ARMEmitter::Reg::r30) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'010 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
|
||||
// Unconditional branch immediate
|
||||
void b(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void b(ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
}
|
||||
else {
|
||||
b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void bl(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void bl(ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
}
|
||||
else {
|
||||
bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Test and branch immediate
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Conditional branch immediate
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Op1 << 24;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Op0 << 4;
|
||||
Instr |= FEXCore::ToUnderlying(Cond);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void UnconditionalBranch(uint32_t Op, FEXCore::ARMEmitter::Register rn) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Encode_rn(rn);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Unconditional branch - immediate
|
||||
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Imm & 0x3FF'FFFF;
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void CompareAndBranch(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Test and branch - immediate
|
||||
void TestAndBranch(uint32_t Op, FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= (Bit >> 5) << 31;
|
||||
Instr |= (Bit & 0b1'1111) << 19;
|
||||
Instr |= (Imm & 0x3FFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
@@ -0,0 +1,100 @@
|
||||
#pragma once
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::ARMEmitter {
|
||||
class Buffer {
|
||||
public:
|
||||
Buffer() {
|
||||
SetBuffer(nullptr, 0);
|
||||
}
|
||||
|
||||
Buffer(uint8_t* Base, uint64_t BaseSize) {
|
||||
SetBuffer(Base, BaseSize);
|
||||
}
|
||||
|
||||
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
|
||||
BufferBase = Base;
|
||||
CurrentOffset = BufferBase;
|
||||
Size = BaseSize;
|
||||
}
|
||||
|
||||
void dc8(uint8_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
|
||||
void dc16(uint16_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
|
||||
void dc32(uint32_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
|
||||
void dc64(uint64_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void EmitString(const char *String) {
|
||||
const auto StringLength = strlen(String);
|
||||
memcpy(CurrentOffset, String, StringLength);
|
||||
CurrentOffset += StringLength;
|
||||
}
|
||||
|
||||
void Align() {
|
||||
// Align the buffer to instruction size
|
||||
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
|
||||
if (!CurrentAlignment) {
|
||||
return;
|
||||
}
|
||||
CurrentOffset += 4 - CurrentAlignment;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T GetCursorAddress() const {
|
||||
return reinterpret_cast<T>(CurrentOffset);
|
||||
}
|
||||
|
||||
static void ClearICache(void* Begin, std::size_t Length) {
|
||||
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
|
||||
}
|
||||
|
||||
size_t GetCursorOffset() const {
|
||||
return static_cast<size_t>(CurrentOffset - BufferBase);
|
||||
}
|
||||
|
||||
uint8_t *GetBufferBase() const {
|
||||
return BufferBase;
|
||||
}
|
||||
|
||||
void CursorIncrement(size_t Size) {
|
||||
CurrentOffset += Size;
|
||||
}
|
||||
|
||||
void SetCursorOffset(size_t Offset) {
|
||||
CurrentOffset = BufferBase + Offset;
|
||||
}
|
||||
|
||||
uint64_t GetBufferSize() const {
|
||||
return Size;
|
||||
}
|
||||
|
||||
protected:
|
||||
|
||||
void ResetBuffer() {
|
||||
CurrentOffset = BufferBase;
|
||||
}
|
||||
|
||||
uint8_t* BufferBase;
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,721 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
/*
|
||||
* Welcome to FEX-Emu's custom AArch64 emitter.
|
||||
* This was written specifically to avoid the performance cost of the vixl emitter.
|
||||
*
|
||||
* There are some specific design constraints in this design to target a couple features:
|
||||
* - High performance
|
||||
* - Low CPU cache performance hit
|
||||
* - Significantly reduced code footprint
|
||||
* - Low number of branches
|
||||
*
|
||||
* These requirements are mostly achieved by removing a bunch of developer conveniences
|
||||
* that vixl provides. The developer needs to take a lot of care to not shoot themselves in the foot.
|
||||
*
|
||||
* Misc design decisions:
|
||||
* - Registers are encoded as basic uint32_t enums.
|
||||
* - Converting between different registers is zero-cost.
|
||||
* - Passing around as arguments are as cheap as registers
|
||||
* - Contrast to vixl where every register requires living on the stack.
|
||||
* - Registers can get encoded in to instructions with a simple `BFM` instruction.
|
||||
*
|
||||
* - Instructions are very simply emitted, allowing direct inlining most of the time.
|
||||
* - These are simple enough that multiple back-to-back instructions get optimized to 128-bit load-store operations.
|
||||
* - Contrast to vixl where pretty much no instruction emitter gets inlined.
|
||||
*
|
||||
* - Instruction emitters are /mostly/ unsized. Most instructions take a size argument first, which gets encoded
|
||||
* directly in to the instruction.
|
||||
* - Contrast to vixl where the register arguments are how the instructions determine operating size.
|
||||
* - Size argument allows FEX to use `CSEL` to select a size at runtime, instead of branching.
|
||||
* - Some instructions are explicitly sized based on register type. Read comments in the respective `inl` files to
|
||||
* see why.
|
||||
* Some scalar/vector operations are an example of this.
|
||||
*
|
||||
* - Almost zero helper functions.
|
||||
* - Primary exception to this rule is load-store operations. These will use a helper to make
|
||||
* it easier to select the correct load-store instruction. Mostly because these are a nightmare selecting
|
||||
* the right instruction.
|
||||
*/
|
||||
namespace FEXCore::ARMEmitter {
|
||||
/*
|
||||
* This `Size` enum is used for most ALU operations.
|
||||
* These follow the AArch64 encoding style in most cases.
|
||||
*/
|
||||
enum class Size : uint32_t {
|
||||
i32Bit = 0,
|
||||
i64Bit,
|
||||
};
|
||||
|
||||
// This allows us to get the `Size` enum in bits.
|
||||
template<Size size>
|
||||
constexpr size_t RegSizeInBits() {
|
||||
constexpr size_t RegSize[] = {
|
||||
32, 64, 128,
|
||||
};
|
||||
return RegSize[FEXCore::ToUnderlying(size)];
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t RegSizeInBits(Size size) {
|
||||
constexpr size_t RegSize[] = {
|
||||
32, 64, 128,
|
||||
};
|
||||
return RegSize[FEXCore::ToUnderlying(size)];
|
||||
}
|
||||
|
||||
/* This `SubRegSize` enum is used for most ASIMD operations.
|
||||
* These follow the AArch64 encoding style in most cases.
|
||||
*/
|
||||
enum class SubRegSize : uint32_t {
|
||||
i8Bit = 0b00,
|
||||
i16Bit = 0b01,
|
||||
i32Bit = 0b10,
|
||||
i64Bit = 0b11,
|
||||
i128Bit = 0b100,
|
||||
};
|
||||
|
||||
// This allows us to get the `SubRegSize` in bits.
|
||||
template<SubRegSize size>
|
||||
constexpr size_t SubRegSizeInBits() {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t SubRegSizeInBits(SubRegSize size) {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
/* This `ScalarRegSize` enum is used for most scalar float
|
||||
* operations.
|
||||
*
|
||||
* This is specifically duplicated from `SubRegSize` to have strongly
|
||||
* typed functions.
|
||||
*
|
||||
* `ScalarRegSize` specifically doesn't have `i128Bit` because scalar operations
|
||||
* can't operate at 128-bit.
|
||||
*/
|
||||
enum class ScalarRegSize : uint32_t {
|
||||
i8Bit = 0b00,
|
||||
i16Bit = 0b01,
|
||||
i32Bit = 0b10,
|
||||
i64Bit = 0b11,
|
||||
};
|
||||
|
||||
// This allows us to get the `ScalarRegSize` in bits.
|
||||
template<ScalarRegSize size>
|
||||
constexpr size_t ScalarRegSizeInBits() {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t ScalarRegSizeInBits(ScalarRegSize size) {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
/* This `VectorRegSizePair` union allows us to have an overlapping type
|
||||
* to select a scalar operation or a vector depending on which operation
|
||||
* we pass in.
|
||||
* Useful in FEX's vector operations that behave as scalar or vector
|
||||
* depending on various factors. But since the operation will have the sa,e
|
||||
* element size, we want to choose the operation more easily
|
||||
*/
|
||||
union VectorRegSizePair {
|
||||
ScalarRegSize Scalar;
|
||||
SubRegSize Vector;
|
||||
};
|
||||
|
||||
// This allows us to create a `VectorRegSizePair` union.
|
||||
[[maybe_unused]]
|
||||
static inline VectorRegSizePair ToVectorSizePair(SubRegSize size) {
|
||||
return VectorRegSizePair {.Vector = size};
|
||||
}
|
||||
[[maybe_unused]]
|
||||
static inline VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
|
||||
return VectorRegSizePair {.Scalar = size};
|
||||
}
|
||||
|
||||
// This `ShiftType` enum is used for ALU shift-register encoded instructions.
|
||||
enum class ShiftType : uint32_t {
|
||||
LSL = 0,
|
||||
LSR,
|
||||
ASR,
|
||||
ROR,
|
||||
};
|
||||
|
||||
// This `ExtendedType` enum is used for ALU extended-register encoded instructions.
|
||||
enum class ExtendedType : uint32_t {
|
||||
UXTB = 0b000,
|
||||
UXTH = 0b001,
|
||||
UXTW = 0b010,
|
||||
UXTX = 0b011,
|
||||
SXTB = 0b100,
|
||||
SXTH = 0b101,
|
||||
SXTW = 0b110,
|
||||
SXTX = 0b111,
|
||||
LSL_32 = UXTW,
|
||||
LSL_64 = UXTX,
|
||||
};
|
||||
|
||||
// This `Condition` enum is used for various conditional instructions.
|
||||
enum class Condition : uint32_t {
|
||||
// Meaning: Int - Float
|
||||
CC_EQ = 0, // Equal - Equal
|
||||
CC_NE, // Not Eq - Not Eq or unordered
|
||||
CC_CS, // Carry set - Greater than, equal, or unordered
|
||||
CC_CC, // Carry clear - Less than
|
||||
CC_MI, // Minus/Negative - Less than
|
||||
CC_PL, // Plus, positive or zero - GT, equal, or unordered
|
||||
CC_VS, // Overflow - Unordered
|
||||
CC_VC, // No Overflow - Ordered
|
||||
CC_HI, // Unsigned higher - GT, or unordered
|
||||
CC_LS, // Unsigned lower or same - LT or EQ
|
||||
CC_GE, // Signed GT or EQ - GT or EQ
|
||||
CC_LT, // Signed LT - LT or Unordered
|
||||
CC_GT, // Signed GT - GT
|
||||
CC_LE, // Signed LT or EQ - LT, EQ, or Unordered
|
||||
CC_AL, // Always - Always
|
||||
CC_NV, // Always - Always
|
||||
|
||||
// Aliases
|
||||
CC_HS = CC_CS,
|
||||
CC_LO = CC_CC,
|
||||
};
|
||||
|
||||
/*
|
||||
* This `StatusFlags` enum is used for conditional compare encoded instructions.
|
||||
* These directly encode to the `nzcv` flags.
|
||||
*/
|
||||
enum class StatusFlags : uint32_t {
|
||||
None = 0,
|
||||
Flag_V = 0b0001,
|
||||
Flag_C = 0b0010,
|
||||
Flag_Z = 0b0100,
|
||||
Flag_N = 0b1000,
|
||||
|
||||
Flag_NZCV = Flag_N | Flag_Z | Flag_C | Flag_V,
|
||||
};
|
||||
|
||||
|
||||
/*
|
||||
* This `IndexType` enum is used for load-store instructions.
|
||||
* Not all load-store instructions use this, so the user needs to be careful.
|
||||
*/
|
||||
enum class IndexType {
|
||||
POST,
|
||||
OFFSET,
|
||||
PRE,
|
||||
|
||||
UNPRIVILEGED,
|
||||
};
|
||||
|
||||
/* This `SVEMemOperand` class is used for the helper SVE load-store instructions.
|
||||
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
|
||||
*/
|
||||
class SVEMemOperand final {
|
||||
public:
|
||||
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
|
||||
: rn {rn}
|
||||
, MetaType {
|
||||
.ScalarScalarType {
|
||||
.Header = { .MemType = TYPE_SCALAR_SCALAR },
|
||||
.rm = rm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, int32_t imm = 0)
|
||||
: rn {rn}
|
||||
, MetaType {
|
||||
.ScalarImmType {
|
||||
.Header = { .MemType = TYPE_SCALAR_IMM },
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
|
||||
Register rn;
|
||||
enum Type {
|
||||
TYPE_SCALAR_SCALAR,
|
||||
TYPE_SCALAR_IMM,
|
||||
TYPE_SCALAR_VECTOR,
|
||||
TYPE_VECTOR_IMM,
|
||||
};
|
||||
struct HeaderStruct {
|
||||
Type MemType;
|
||||
};
|
||||
|
||||
union {
|
||||
HeaderStruct Header;
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
Register rm;
|
||||
} ScalarScalarType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
int32_t Imm;
|
||||
} ScalarImmType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
ZRegister zm;
|
||||
// TODO: Implement support for modifier
|
||||
} ScalarVectorType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
// rn will be a ZRegister
|
||||
int32_t Imm;
|
||||
} VectorImmType;
|
||||
} MetaType;
|
||||
};
|
||||
|
||||
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
|
||||
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
|
||||
*/
|
||||
class ExtendedMemOperand final {
|
||||
public:
|
||||
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
|
||||
: rn {rn}
|
||||
, MetaType {
|
||||
.ExtendedType {
|
||||
.Header = { .MemType = TYPE_EXTENDED },
|
||||
.rm = rm,
|
||||
.Option = Option,
|
||||
.Shift = Shift,
|
||||
}
|
||||
} {}
|
||||
ExtendedMemOperand(XRegister rn, IndexType Index = IndexType::OFFSET, int32_t Imm = 0)
|
||||
: rn {rn}
|
||||
, MetaType {
|
||||
.ImmType {
|
||||
.Header = { .MemType = TYPE_IMM },
|
||||
.Index = Index,
|
||||
.Imm = Imm,
|
||||
}
|
||||
} {}
|
||||
|
||||
Register rn;
|
||||
enum Type {
|
||||
TYPE_EXTENDED,
|
||||
TYPE_IMM,
|
||||
};
|
||||
struct HeaderStruct {
|
||||
Type MemType;
|
||||
};
|
||||
union {
|
||||
HeaderStruct Header;
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
Register rm;
|
||||
ExtendedType Option;
|
||||
uint32_t Shift;
|
||||
} ExtendedType;
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
IndexType Index;
|
||||
int32_t Imm;
|
||||
} ImmType;
|
||||
} MetaType;
|
||||
};
|
||||
|
||||
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenSystemReg() {
|
||||
return op0 << 19 |
|
||||
op1 << 16 |
|
||||
CRn << 12 |
|
||||
CRm << 8 |
|
||||
op2 << 5;
|
||||
};
|
||||
|
||||
// This `SystemRegister` enum is used for the mrs/msr instructions.
|
||||
enum class SystemRegister : uint32_t {
|
||||
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
|
||||
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
|
||||
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
|
||||
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
|
||||
};
|
||||
|
||||
template<uint32_t op1, uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenDCReg() {
|
||||
return op1 << 16 |
|
||||
CRm << 8 |
|
||||
op2 << 5;
|
||||
};
|
||||
|
||||
// This `DataCacheOperation` enum is used for the dc instruction.
|
||||
enum class DataCacheOperation : uint32_t {
|
||||
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
|
||||
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
|
||||
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
|
||||
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
|
||||
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
|
||||
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
|
||||
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
|
||||
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
|
||||
|
||||
// MTE2
|
||||
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
|
||||
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
|
||||
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
|
||||
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
|
||||
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
|
||||
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
|
||||
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
|
||||
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
|
||||
|
||||
// MTE
|
||||
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
|
||||
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
|
||||
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
|
||||
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
|
||||
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
|
||||
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
|
||||
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
|
||||
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
|
||||
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
|
||||
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
|
||||
|
||||
// DPB
|
||||
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
|
||||
|
||||
// DPB2
|
||||
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
|
||||
};
|
||||
|
||||
template<uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenHintBarrierReg() {
|
||||
return CRm << 8 |
|
||||
op2 << 5;
|
||||
}
|
||||
|
||||
// This `HintRegister` enum is used for the hint instruction.
|
||||
enum class HintRegister : uint32_t {
|
||||
NOP = GenHintBarrierReg<0b0000, 0b000>(),
|
||||
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
|
||||
WFE = GenHintBarrierReg<0b0000, 0b010>(),
|
||||
WFI = GenHintBarrierReg<0b0000, 0b011>(),
|
||||
SEV = GenHintBarrierReg<0b0000, 0b100>(),
|
||||
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
|
||||
DGH = GenHintBarrierReg<0b0000, 0b110>(),
|
||||
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
|
||||
};
|
||||
|
||||
// This `BarrierRegister` enum is used for the various barrier instructions.
|
||||
enum class BarrierRegister : uint32_t {
|
||||
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
|
||||
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
|
||||
DSB = GenHintBarrierReg<0b0000, 0b100>(),
|
||||
DMB = GenHintBarrierReg<0b0000, 0b101>(),
|
||||
ISB = GenHintBarrierReg<0b0000, 0b110>(),
|
||||
SB = GenHintBarrierReg<0b0000, 0b111>(),
|
||||
};
|
||||
|
||||
// This `BarrierScope` enum is used for the dsb/dmb instructions.
|
||||
enum class BarrierScope : uint32_t {
|
||||
// Outer shareable
|
||||
OSHLD = 0b0001,
|
||||
OSHST = 0b0010,
|
||||
OSH = 0b0011,
|
||||
// Non shareable
|
||||
NSHLD = 0b0101,
|
||||
NSHST = 0b0110,
|
||||
NSH = 0b0111,
|
||||
// Inner shareable
|
||||
ISHLD = 0b1001,
|
||||
ISHST = 0b1010,
|
||||
ISH = 0b1011,
|
||||
// Full System visibility
|
||||
LD = 0b1101,
|
||||
ST = 0b1110,
|
||||
SY = 0b1111,
|
||||
};
|
||||
|
||||
// This `Prefetch` enum is used for prefetch instructions.
|
||||
enum class Prefetch : uint32_t {
|
||||
// Prefetch for load
|
||||
PLDL1KEEP = 0b00000,
|
||||
PLDL1STRM = 0b00001,
|
||||
PLDL2KEEP = 0b00010,
|
||||
PLDL2STRM = 0b00011,
|
||||
PLDL3KEEP = 0b00100,
|
||||
PLDL3STRM = 0b00101,
|
||||
|
||||
// Preload instructions
|
||||
PLIL1KEEP = 0b01000,
|
||||
PLIL1STRM = 0b01001,
|
||||
PLIL2KEEP = 0b01010,
|
||||
PLIL2STRM = 0b01011,
|
||||
PLIL3KEEP = 0b01100,
|
||||
PLIL3STRM = 0b01101,
|
||||
|
||||
// Preload for store
|
||||
PSTL1KEEP = 0b10000,
|
||||
PSTL1STRM = 0b10001,
|
||||
PSTL2KEEP = 0b10010,
|
||||
PSTL2STRM = 0b10011,
|
||||
PSTL3KEEP = 0b10100,
|
||||
PSTL3STRM = 0b10101,
|
||||
};
|
||||
|
||||
// This `PredicatePattern` enun is used for some SVE instructions.
|
||||
enum class PredicatePattern : uint32_t {
|
||||
SVE_POW2 = 0b00000,
|
||||
SVE_VL1 = 0b00001,
|
||||
SVE_VL2 = 0b00010,
|
||||
SVE_VL3 = 0b00011,
|
||||
SVE_VL4 = 0b00100,
|
||||
SVE_VL5 = 0b00101,
|
||||
SVE_VL6 = 0b00110,
|
||||
SVE_VL7 = 0b00111,
|
||||
SVE_VL8 = 0b01000,
|
||||
SVE_VL16 = 0b01001,
|
||||
SVE_VL32 = 0b01010,
|
||||
SVE_VL64 = 0b01011,
|
||||
SVE_VL128 = 0b01100,
|
||||
SVE_VL256 = 0b01101,
|
||||
SVE_MUL4 = 0b11101,
|
||||
SVE_MUL3 = 0b11110,
|
||||
SVE_ALL = 0b11111,
|
||||
};
|
||||
|
||||
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `below` an instruction that uses it.
|
||||
* Which means that a branch would jump backwards.
|
||||
*/
|
||||
struct BackwardLabel {
|
||||
uint8_t *Location{};
|
||||
};
|
||||
|
||||
/* This `ForwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `above` an instruction that uses it.
|
||||
* Which means that a branch would jump forwards.
|
||||
*
|
||||
* This can be bound to multiple instructions, so it needs a vector for each bind instruction type.
|
||||
*/
|
||||
struct ForwardLabel {
|
||||
struct Instructions {
|
||||
enum class InstType {
|
||||
ADR,
|
||||
ADRP,
|
||||
B,
|
||||
BC,
|
||||
TEST_BRANCH,
|
||||
RELATIVE_LOAD,
|
||||
};
|
||||
uint8_t *Location{};
|
||||
InstType Type;
|
||||
};
|
||||
std::vector<Instructions> Insts{};
|
||||
};
|
||||
|
||||
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is in either direction of an instruction that uses it.
|
||||
* Which means a branch could jump backwards or forwards depending on situation.
|
||||
*/
|
||||
struct BiDirectionalLabel {
|
||||
BackwardLabel Backward;
|
||||
ForwardLabel Forward;
|
||||
};
|
||||
|
||||
// This is an emitter that is designed around the smallest code bloat as possible.
|
||||
// Eschewing most developer convenience in order to keep code as small as possible.
|
||||
|
||||
// Choices:
|
||||
// - Size of ops passed as an argument rather than template to let the compiler use csel instead of branching.
|
||||
// - Registers are unsized so they can be passed in a GPR and not need conversion operations
|
||||
class Emitter : public FEXCore::ARMEmitter::Buffer {
|
||||
public:
|
||||
Emitter() = default;
|
||||
|
||||
Emitter(uint8_t* Base, uint64_t BaseSize)
|
||||
: Buffer (Base, BaseSize) {
|
||||
}
|
||||
|
||||
// Bind a backward label to an address.
|
||||
// Address that is bound is the current emitter location.
|
||||
void Bind(BackwardLabel *Label) {
|
||||
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
}
|
||||
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
template<bool WarnAboutEmpty = false>
|
||||
void Bind(ForwardLabel *Label) {
|
||||
if constexpr (WarnAboutEmpty) {
|
||||
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
|
||||
}
|
||||
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
for (const auto &Inst : Label->Insts) {
|
||||
// Patch up the instructions
|
||||
switch (Inst.Type) {
|
||||
case ForwardLabel::Instructions::InstType::ADR: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575, "Unscaled offset too large");
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= (Offset & 0b11) << 29;
|
||||
Inst |= (Offset >> 2) << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::Instructions::InstType::ADRP: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -4294967296 && Imm <= 4294963200 && (Imm & 0xFFF) == 0, "Unscaled offset too large");
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= (Offset & 0b11) << 29;
|
||||
Inst |= (Offset >> 2) << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::Instructions::InstType::B: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FF'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= Offset;
|
||||
*Instruction = Inst;
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::Instructions::InstType::TEST_BRANCH: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~(InstMask << 5);
|
||||
Inst |= Offset << 5;
|
||||
*Instruction = Inst;
|
||||
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::Instructions::InstType::BC:
|
||||
case ForwardLabel::Instructions::InstType::RELATIVE_LOAD: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x7'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~(InstMask << 5);
|
||||
Inst |= Offset << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Bind a bidirectional location to a location.
|
||||
// Binds both forwards and backwards depending on how the label was used.
|
||||
void Bind(BiDirectionalLabel *Label) {
|
||||
if (!Label->Backward.Location) {
|
||||
Bind(&Label->Backward);
|
||||
}
|
||||
Bind<false>(&Label->Forward);
|
||||
}
|
||||
|
||||
public:
|
||||
// TODO: Implement SME when it matters.
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/ALUOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/BranchOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/LoadstoreOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/SystemOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/ScalarOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/ASIMDOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/SVEOps.inl"
|
||||
|
||||
private:
|
||||
template<typename T>
|
||||
uint32_t Encode_ra(T Reg) const {
|
||||
return Reg.Idx() << 10;
|
||||
}
|
||||
uint32_t Encode_ra(uint32_t Reg) const {
|
||||
return Reg << 10;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_rt2(T Reg) const {
|
||||
return Reg.Idx() << 10;
|
||||
}
|
||||
template<>
|
||||
uint32_t Encode_rt2(uint32_t Reg) const {
|
||||
return Reg << 10;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_rm(T Reg) const {
|
||||
return Reg.Idx() << 16;
|
||||
}
|
||||
uint32_t Encode_rm(uint32_t Reg) const {
|
||||
return Reg << 16;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_rs(T Reg) const {
|
||||
return Reg.Idx() << 16;
|
||||
}
|
||||
uint32_t Encode_rs(uint32_t Reg) const {
|
||||
return Reg << 16;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_rn(T Reg) const {
|
||||
return Reg.Idx() << 5;
|
||||
}
|
||||
uint32_t Encode_rn(uint32_t Reg) const {
|
||||
return Reg << 5;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_rd(T Reg) const {
|
||||
return Reg.Idx();
|
||||
}
|
||||
uint32_t Encode_rd(uint32_t Reg) const {
|
||||
return Reg;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_rt(T Reg) const {
|
||||
return Reg.Idx();
|
||||
}
|
||||
template<>
|
||||
uint32_t Encode_rt(Prefetch Reg) const {
|
||||
return FEXCore::ToUnderlying(Reg);
|
||||
}
|
||||
uint32_t Encode_rt(uint32_t Reg) const {
|
||||
return Reg;
|
||||
}
|
||||
template<typename T>
|
||||
uint32_t Encode_pd(T Reg) const {
|
||||
return FEXCore::ToUnderlying(Reg);
|
||||
}
|
||||
};
|
||||
}
|
||||
+4906
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
+1793
File diff suppressed because it is too large.
Load diff
+175
@@ -0,0 +1,175 @@
|
||||
/* System instruction emitters.
|
||||
*
|
||||
* This is mostly a mashup of various instruction types.
|
||||
* Nothing follows an explicit pattern since they are mostly different.
|
||||
*/
|
||||
public:
|
||||
// System with result
|
||||
// TODO: SYSL
|
||||
// System Instruction
|
||||
// TODO: AT
|
||||
// TODO: CFP
|
||||
// TODO: CPP
|
||||
void dc(FEXCore::ARMEmitter::DataCacheOperation DCOp, FEXCore::ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
|
||||
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
|
||||
}
|
||||
// TODO: DVP
|
||||
// TODO: IC
|
||||
// TODO: TLBI
|
||||
|
||||
// Exception generation
|
||||
void svc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
|
||||
}
|
||||
void hvc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
|
||||
}
|
||||
void smc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
|
||||
}
|
||||
void brk(uint32_t Imm) {
|
||||
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
|
||||
}
|
||||
void hlt(uint32_t Imm) {
|
||||
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
|
||||
}
|
||||
void tcancel(uint32_t Imm) {
|
||||
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
|
||||
}
|
||||
void dcps1(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
|
||||
}
|
||||
void dcps2(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
|
||||
}
|
||||
void dcps3(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
|
||||
}
|
||||
// System instructions with register argument
|
||||
void wfet(FEXCore::ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b000, rt);
|
||||
}
|
||||
void wfit(FEXCore::ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b001, rt);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void nop() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::NOP);
|
||||
}
|
||||
void yield() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::YIELD);
|
||||
}
|
||||
void wfe() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::WFE);
|
||||
}
|
||||
void wfi() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::WFI);
|
||||
}
|
||||
void sev() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::SEV);
|
||||
}
|
||||
void sevl() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::SEVL);
|
||||
}
|
||||
void dgh() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::DGH);
|
||||
}
|
||||
void csdb() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::CSDB);
|
||||
}
|
||||
|
||||
// Barriers
|
||||
void clrex(uint32_t imm = 15) {
|
||||
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
}
|
||||
void dsb(FEXCore::ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void dmb(FEXCore::ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void isb() {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(FEXCore::ARMEmitter::BarrierScope::SY));
|
||||
}
|
||||
void sb() {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::SB, 0);
|
||||
}
|
||||
void tcommit() {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void msr(FEXCore::ARMEmitter::SystemRegister reg, FEXCore::ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
|
||||
SystemRegisterMove(Op, rt, reg);
|
||||
}
|
||||
|
||||
void mrs(FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::SystemRegister reg) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
|
||||
SystemRegisterMove(Op, rd, reg);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
// Exception Generation
|
||||
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
|
||||
LOGMAN_THROW_AA_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
|
||||
|
||||
uint32_t Instr = 0b1101'0100 << 24;
|
||||
|
||||
Instr |= opc << 21;
|
||||
Instr |= Imm << 5;
|
||||
Instr |= op2 << 2;
|
||||
Instr |= LL;
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System instructions with register argument
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, FEXCore::ARMEmitter::Register rt) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
|
||||
|
||||
Instr |= CRm << 8;
|
||||
Instr |= op2 << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void Hint(FEXCore::ARMEmitter::HintRegister Reg) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Barriers
|
||||
void Barrier(FEXCore::ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
|
||||
Instr |= CRm << 8;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System Instruction
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, FEXCore::ARMEmitter::Register rt) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= L << 21;
|
||||
Instr |= SubOp;
|
||||
Instr |= Encode_rt(rt);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void SystemRegisterMove(uint32_t Op, FEXCore::ARMEmitter::Register rt, FEXCore::ARMEmitter::SystemRegister reg) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= FEXCore::ToUnderlying(reg);
|
||||
Instr |= Encode_rt(rt);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
|
||||
#include <signal.h>
|
||||
#include <string.h>
|
||||
@@ -10,7 +11,6 @@
|
||||
#include <stdint.h>
|
||||
#include <type_traits>
|
||||
|
||||
|
||||
namespace FEXCore::ArchHelpers::Context {
|
||||
|
||||
enum ContextFlags : uint32_t {
|
||||
@@ -76,6 +76,7 @@ static inline mcontext_t* GetMContext(void* ucontext) {
|
||||
#ifdef _M_ARM_64
|
||||
|
||||
constexpr uint32_t FPR_MAGIC = 0x46508001U;
|
||||
constexpr uint32_t ESR1_MAGIC = 0x45535201U;
|
||||
|
||||
struct HostCTXHeader {
|
||||
uint32_t Magic;
|
||||
@@ -89,6 +90,11 @@ struct HostFPRState {
|
||||
__uint128_t FPRs[32];
|
||||
};
|
||||
|
||||
struct HostESRState {
|
||||
HostCTXHeader Head;
|
||||
uint64_t ESR;
|
||||
};
|
||||
|
||||
static inline uint64_t GetSp(void* ucontext) {
|
||||
return GetMContext(ucontext)->sp;
|
||||
}
|
||||
@@ -129,6 +135,61 @@ static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmESR(void* ucontext) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
|
||||
size_t i = 0;
|
||||
auto HostState = reinterpret_cast<HostCTXHeader*>(&MContext->__reserved[i]);
|
||||
do {
|
||||
if (HostState->Magic == ESR1_MAGIC) {
|
||||
auto ESR = reinterpret_cast<HostESRState*>(HostState);
|
||||
return ESR->ESR;
|
||||
}
|
||||
i += HostState->Size;
|
||||
HostState = reinterpret_cast<HostCTXHeader*>(&MContext->__reserved[i]);
|
||||
} while (HostState->Size != 0);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
constexpr static uint64_t ESR1_EC = 0b111111U << 26;
|
||||
constexpr static uint64_t ESR1_EC_DataAbort = 0b100100U << 26;
|
||||
|
||||
// Write-Not-Read flag
|
||||
// When set - Abort is due to a write
|
||||
constexpr static uint64_t ESR1_WNR = 1 << 6;
|
||||
|
||||
// DFSC - Default Status Code
|
||||
// Translation fault - No page mapped
|
||||
// Permissions fault - Page mapped but with incorrect permission from access.
|
||||
constexpr static uint64_t ESR1_DataAbort_DFSC = 0b111111;
|
||||
constexpr static uint64_t ESR1_DataAbort_TranslationFault_EL0 = 0b000111;
|
||||
constexpr static uint64_t ESR1_DataAbort_PermissionFault_EL0 = 0b001111;
|
||||
constexpr static uint64_t ESR1_DataAbort_Level = 0b11;
|
||||
constexpr static uint64_t ESR1_DataAbort_Level_EL3 = 0b00;
|
||||
constexpr static uint64_t ESR1_DataAbort_Level_EL2 = 0b01;
|
||||
constexpr static uint64_t ESR1_DataAbort_Level_EL1 = 0b10;
|
||||
constexpr static uint64_t ESR1_DataAbort_Level_EL0 = 0b11;
|
||||
|
||||
static inline uint32_t GetProtectFlags(void* ucontext) {
|
||||
uint64_t ESR = GetArmESR(ucontext);
|
||||
LOGMAN_THROW_A_FMT((ESR & ESR1_EC) == ESR1_EC_DataAbort, "Unknown ESR1 EC type: 0x{:x} != 0x{:x}", ESR & ESR1_EC, ESR1_EC_DataAbort);
|
||||
|
||||
uint32_t ProtectFlags{};
|
||||
if ((ESR & ESR1_DataAbort_Level) == ESR1_DataAbort_Level_EL0) {
|
||||
// Always a user error for us.
|
||||
ProtectFlags |= X86State::X86_PF_USER;
|
||||
}
|
||||
|
||||
if (ESR & ESR1_WNR) {
|
||||
// Fault was due to a write
|
||||
ProtectFlags |= X86State::X86_PF_WRITE;
|
||||
}
|
||||
|
||||
// PF_PROT is not returned to user on x86, so don't return the difference between permission fault and translation fault.
|
||||
return ProtectFlags;
|
||||
}
|
||||
|
||||
using ContextBackup = ArmContextBackup;
|
||||
template <typename T>
|
||||
static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
@@ -222,6 +283,10 @@ static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint32_t GetProtectFlags(void* ucontext) {
|
||||
return GetMContext(ucontext)->gregs[REG_ERR];
|
||||
}
|
||||
|
||||
using ContextBackup = X86ContextBackup;
|
||||
template <typename T>
|
||||
static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
+1
-1
@@ -421,7 +421,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(1 << 1) | // PCLMULQDQ
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(0 << 4) | // DS-CPL
|
||||
|
||||
+35
-28
@@ -44,6 +44,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
@@ -221,7 +222,7 @@ namespace FEXCore::Context {
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
|
||||
#else
|
||||
@@ -238,16 +239,16 @@ namespace FEXCore::Context {
|
||||
|
||||
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
|
||||
|
||||
#if (_M_X86_64)
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#elif (_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateArm64(this, DispatcherConfig);
|
||||
#elif JIT_X86_64
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
|
||||
#endif
|
||||
|
||||
// Initialize common signal handlers
|
||||
|
||||
|
||||
auto PauseHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
|
||||
};
|
||||
@@ -573,7 +574,7 @@ namespace FEXCore::Context {
|
||||
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, Thread);
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
@@ -674,6 +675,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
@@ -740,7 +743,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
@@ -749,7 +754,7 @@ namespace FEXCore::Context {
|
||||
|
||||
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
|
||||
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
if (Handler != CustomIRHandlers.end()) {
|
||||
TotalInstructions = 1;
|
||||
@@ -851,28 +856,31 @@ namespace FEXCore::Context {
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
}
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError) {
|
||||
if (TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
}
|
||||
else {
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
const bool NeedsBlockEnd = (HadDispatchError && TotalInstructions > 0) ||
|
||||
(Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock);
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
break;
|
||||
}
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
if (Thread->OpDispatcher->FinishOp(DecodedInfo->PC + DecodedInfo->InstSize, i + 1 == InstsInBlock)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
@@ -1011,6 +1019,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
@@ -1182,7 +1191,7 @@ namespace FEXCore::Context {
|
||||
|
||||
static void InvalidateGuestCodeRangeInternal(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(CTX->ThreadCreationMutex);
|
||||
|
||||
|
||||
for (auto &Thread : CTX->Threads) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
@@ -1190,7 +1199,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
}
|
||||
|
||||
@@ -1206,8 +1215,6 @@ namespace FEXCore::Context {
|
||||
IsMemoryShared = true;
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
LogMan::Msg::IFmt("Migrating to shared memory mode");
|
||||
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
LogMan::Throw::AFmt(Threads.size() == 1, "First MarkMemoryShared called must be before creating any threads");
|
||||
|
||||
@@ -1233,9 +1240,9 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(Thread->CTX->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->DebugStore.erase(GuestRIP);
|
||||
|
||||
+228
-183
@@ -1,3 +1,4 @@
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
@@ -30,20 +31,28 @@
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SetAllowAssembler(true);
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&Decoder}
|
||||
#endif
|
||||
{
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode a 256-bit vector width if we are running in the simulator.
|
||||
Simulator.SetVectorLengthInBits(256);
|
||||
#endif
|
||||
|
||||
EmitDispatcher();
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::EmitDispatcher() {
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
#endif
|
||||
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
|
||||
@@ -55,9 +64,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// Ptr();
|
||||
// }
|
||||
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
Literal l_CompileBlock {GetCompileBlockPtr()};
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::ForwardLabel l_Sleep;
|
||||
ARMEmitter::ForwardLabel l_CompileBlock;
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -65,12 +74,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// Push our memory base to the correct register
|
||||
// Move our thread pointer to the correct register
|
||||
// This is passed in to parameter 0 (x0)
|
||||
mov(STATE, x0);
|
||||
mov(STATE, ARMEmitter::XReg::x0);
|
||||
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
add(x0, sp, 0);
|
||||
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
|
||||
str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -80,91 +89,97 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
aarch64::Label FullLookup{};
|
||||
aarch64::Label CallBlock{};
|
||||
aarch64::Label LoopTop{};
|
||||
aarch64::Label ExitSpillSRA{};
|
||||
aarch64::Label ThreadPauseHandler{};
|
||||
ARMEmitter::BiDirectionalLabel FullLookup{};
|
||||
ARMEmitter::BiDirectionalLabel CallBlock{};
|
||||
ARMEmitter::BackwardLabel LoopTop{};
|
||||
|
||||
bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetLabelAddress<uint64_t>(&LoopTop);
|
||||
Bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify x2 since it contains our RIP once the block doesn't exist
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
|
||||
auto RipReg = x2;
|
||||
|
||||
auto RipReg = ARMEmitter::XReg::x2;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
ldp(x3, x0, MemOperand(x0));
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
|
||||
br(x3);
|
||||
br(ARMEmitter::Reg::r3);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
bind(&FullLookup);
|
||||
Bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, VirtualMemorySize - 1);
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), VirtualMemorySize - 1);
|
||||
}
|
||||
else {
|
||||
LoadConstant(x3, VirtualMemorySize);
|
||||
and_(x3, RipReg, x3);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
aarch64::Label NoBlock;
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(x1, x3, 12);
|
||||
lsr(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 12);
|
||||
|
||||
// Load the pointer from the offset
|
||||
ldr(x0, MemOperand(x0, x1, Shift::LSL, 3));
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(x0, &NoBlock);
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(x1, x3, 0x0FFF);
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(x0, x0, Operand(x1, Shift::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry))));
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// Load the guest address first to ensure it maps to the address we are currently at
|
||||
// This fixes aliasing problems
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)));
|
||||
cmp(x1, RipReg);
|
||||
b(&NoBlock, Condition::ne);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode));
|
||||
cmp(ARMEmitter::XReg::x1, RipReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NoBlock);
|
||||
|
||||
// Now load the actual host block to execute if we can
|
||||
ldr(x3, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)));
|
||||
cbz(x3, &NoBlock);
|
||||
ldr(ARMEmitter::XReg::x3, ARMEmitter::Reg::r0, offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode));
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x2, ARMEmitter::Reg::r0);
|
||||
|
||||
// Jump to the block
|
||||
br(x3);
|
||||
br(ARMEmitter::Reg::r3);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
@@ -178,7 +193,6 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
@@ -193,48 +207,52 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(x0, ~0ULL);
|
||||
stp(x0, x0, MemOperand(sp, -16, PreIndex));
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
add(x2, sp, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
blr(x3);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(x4, x0);
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
LoadConstant(x2, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
mov(ARMEmitter::XReg::x4, ARMEmitter::XReg::x0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(sp, sp, 16);
|
||||
|
||||
mov(x0, x4);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
mov(ARMEmitter::XReg::x0, ARMEmitter::XReg::x4);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
br(x0);
|
||||
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
bind(&NoBlock);
|
||||
Bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
@@ -248,39 +266,42 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(x0, ~0ULL);
|
||||
stp(x0, x2, MemOperand(sp, -16, PreIndex));
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
add(x2, sp, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(x2, MemOperand(sp, 8, Offset));
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, 8);
|
||||
}
|
||||
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x3, &l_CompileBlock);
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x3, &l_CompileBlock);
|
||||
|
||||
// X2 contains our guest RIP
|
||||
blr(x3); // { CTX, Frame, RIP}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
LoadConstant(x2, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(sp, sp, 16);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
@@ -300,7 +321,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
@@ -331,8 +352,8 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(x1, 0);
|
||||
ldr(x1, MemOperand(x1));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -340,16 +361,19 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
bind(&ThreadPauseHandler);
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
// We will have faulted and jumped to this location at this point
|
||||
|
||||
// Call our sleep handler
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x2, &l_Sleep);
|
||||
blr(x2);
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, &l_Sleep);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void *, void *>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PauseReturnInstruction = GetCursorAddress<uint64_t>();
|
||||
// Fault to start running again
|
||||
@@ -378,27 +402,27 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
PushCalleeSavedRegisters();
|
||||
|
||||
// First thing we need to move the thread state pointer back in to our register
|
||||
mov(STATE, x0);
|
||||
mov(STATE, ARMEmitter::XReg::x0);
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
add(w2, w2, 1);
|
||||
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
ldr(ARMEmitter::WReg::w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
add(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 1);
|
||||
str(ARMEmitter::WReg::w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(x2, x2, 16);
|
||||
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 16);
|
||||
str(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
str(x0, MemOperand(x2));
|
||||
str(ARMEmitter::XReg::x0, ARMEmitter::Reg::r2, 0);
|
||||
|
||||
// Store RIP to the context state
|
||||
str(x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
str(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// load static regs
|
||||
if (config.StaticRegisterAllocation)
|
||||
@@ -411,12 +435,15 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
{
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -430,12 +457,15 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
{
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -449,12 +479,15 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
{
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -468,12 +501,16 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
{
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -484,16 +521,16 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
place(&l_CTX);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
Bind(&l_CTX);
|
||||
dc64(reinterpret_cast<uintptr_t>(CTX));
|
||||
Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
dc64(GetCompileBlockPtr());
|
||||
|
||||
|
||||
FinalizeCode();
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
GetBuffer()->SetExecutable();
|
||||
ClearICache(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
@@ -502,99 +539,107 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
}
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void Arm64Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(DispatchPtr));
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.WriteXRegister(1, RIP);
|
||||
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(CallbackPtr));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxGDBPauseCheckSize};
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
ARMEmitter::ForwardLabel RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
|
||||
emit.ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
emit.ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Thread));
|
||||
emit.ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, offsetof(FEXCore::Core::InternalThreadState, CTX)); // Get Context
|
||||
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::Context, Config.RunningMode));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(w0, &RunBlock);
|
||||
emit.cbz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, &RunBlock);
|
||||
{
|
||||
Literal l_GuestRIP {GuestRIP};
|
||||
ARMEmitter::ForwardLabel l_GuestRIP;
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(x0, &l_GuestRIP);
|
||||
emit.str(x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
emit.ldr(ARMEmitter::XReg::x0, &l_GuestRIP);
|
||||
emit.str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(x0);
|
||||
emit.place(&l_GuestRIP);
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(ARMEmitter::Reg::r0);
|
||||
emit.Bind(&l_GuestRIP);
|
||||
emit.dc64(GuestRIP);
|
||||
}
|
||||
emit.bind(&RunBlock);
|
||||
emit.FinalizeCode();
|
||||
emit.Bind(&RunBlock);
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
auto UsedBytes = emit.GetCursorOffset();
|
||||
emit.ClearICache(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxInterpreterTrampolineSize};
|
||||
ARMEmitter::ForwardLabel InlineIRData;
|
||||
|
||||
aarch64::Label InlineIRData;
|
||||
emit.mov(ARMEmitter::XReg::x0, STATE);
|
||||
emit.adr(ARMEmitter::Reg::r1, &InlineIRData);
|
||||
|
||||
emit.mov(x0, STATE);
|
||||
emit.adr(x1, &InlineIRData);
|
||||
emit.ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
emit.blr(ARMEmitter::Reg::r3);
|
||||
|
||||
emit.ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
emit.blr(x3);
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
emit.br(ARMEmitter::Reg::r0);
|
||||
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
emit.br(x0);
|
||||
emit.Bind(&InlineIRData);
|
||||
|
||||
emit.bind(&InlineIRData);
|
||||
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
auto UsedBytes = emit.GetCursorOffset();
|
||||
emit.ClearICache(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
|
||||
if (IgnoreMask & (1U << SRA64[i].Idx())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].Idx());
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -11,6 +15,9 @@ namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
@@ -20,6 +27,13 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) override;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) override;
|
||||
#endif
|
||||
|
||||
void EmitDispatcher();
|
||||
|
||||
protected:
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
@@ -29,6 +43,11 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
vixl::aarch64::Simulator Simulator;
|
||||
#endif
|
||||
};
|
||||
|
||||
}
|
||||
@@ -212,12 +212,20 @@ void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread,
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
@@ -280,15 +288,14 @@ static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t *HostSigInfo) {
|
||||
return Signal;
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
|
||||
static uint32_t ConvertSignalToError(void *ucontext, int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
// Always a user fault for us
|
||||
// XXX: PF_PROT and PF_WRITE
|
||||
return X86State::X86_PF_USER;
|
||||
return ArchHelpers::Context::GetProtectFlags(ucontext);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -469,7 +476,7 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
|
||||
@@ -565,10 +572,13 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
@@ -578,13 +588,11 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
|
||||
@@ -32,7 +32,7 @@ struct DispatcherConfig {
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -75,12 +75,12 @@ public:
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
|
||||
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
virtual void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
|
||||
+60
-14
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
@@ -450,8 +451,17 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DestSize = 2;
|
||||
}
|
||||
else if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_128BIT);
|
||||
DestSize = 16;
|
||||
if (Options.L) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_256BIT);
|
||||
DestSize = 32;
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_128BIT);
|
||||
DestSize = 16;
|
||||
}
|
||||
}
|
||||
else if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_256BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_256BIT);
|
||||
DestSize = 32;
|
||||
}
|
||||
else if (HasNarrowingDisplacement &&
|
||||
(DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_DEF ||
|
||||
@@ -482,7 +492,14 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
|
||||
if (Options.L) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
|
||||
}
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_256BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
|
||||
}
|
||||
else if (HasNarrowingDisplacement &&
|
||||
(SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_DEF ||
|
||||
@@ -776,6 +793,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
if (Op == 0xC5) { // Two byte VEX
|
||||
pp = Byte1 & 0b11;
|
||||
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
|
||||
options.L = (Byte1 & 0b100) != 0;
|
||||
}
|
||||
else { // 0xC4 = Three byte VEX
|
||||
const uint8_t Byte2 = ReadByte();
|
||||
@@ -783,6 +801,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
map_select = Byte1 & 0b11111;
|
||||
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
options.L = (Byte2 & 0b100) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
@@ -1112,6 +1131,35 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
if (FinalInstruction) {
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t TargetRIP = 0;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
if (DecodeInst->OP == 0xE8) { // Call - immediate target
|
||||
const uint64_t NextRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
|
||||
if (GPRSize == 4) {
|
||||
// If we are running a 32bit guest then wrap around addresses that go above 32bit
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
if (TargetRIP == NextRIP) {
|
||||
// Optimize the case that the instruction is jumping just after itself.
|
||||
// This is a GOT calculation which we can optimize out.
|
||||
// Optimization occurs inside of the OpDispatcher implementation
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
@@ -1132,6 +1180,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1166,7 +1215,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
@@ -1236,23 +1284,21 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
CanContinue = true;
|
||||
}
|
||||
|
||||
bool FinalInstruction = DecodedSize >= CTX->Config.MaxInstPerBlock ||
|
||||
DecodedSize >= DefaultDecodedBufferSize ||
|
||||
TotalInstructions >= CTX->Config.MaxInstPerBlock;
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
BranchTargetInMultiblockRange();
|
||||
|
||||
// Bypass branches if we can continue through them in some cases.
|
||||
CanContinue |= BranchTargetCanContinue(FinalInstruction);
|
||||
}
|
||||
|
||||
if (!CanContinue) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (DecodedSize >= CTX->Config.MaxInstPerBlock ||
|
||||
DecodedSize >= DefaultDecodedBufferSize) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (TotalInstructions >= CTX->Config.MaxInstPerBlock) {
|
||||
if (FinalInstruction || !CanContinue) {
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -49,6 +49,7 @@ private:
|
||||
struct DecodedHeader {
|
||||
uint8_t vvvv; // Encoded operand in a VEX prefix.
|
||||
bool w; // VEX.W bit.
|
||||
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
|
||||
};
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
@@ -57,6 +58,7 @@ private:
|
||||
bool DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
bool BranchTargetCanContinue(bool FinalInstruction) const;
|
||||
|
||||
uint8_t ReadByte();
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
|
||||
+52
-33
@@ -1,7 +1,7 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
@@ -50,8 +50,12 @@ static uint32_t GetDCZID() {
|
||||
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
@@ -61,15 +65,26 @@ HostFeatures::HostFeatures() {
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
SupportsAVX = true;
|
||||
#else
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
@@ -79,37 +94,6 @@ HostFeatures::HostFeatures() {
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
#endif
|
||||
#ifdef _M_X86_64
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
@@ -130,6 +114,40 @@ HostFeatures::HostFeatures() {
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
#endif
|
||||
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
SupportsPMULL_128Bit = Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
@@ -139,5 +157,6 @@ HostFeatures::HostFeatures() {
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
+35
-17
@@ -894,33 +894,51 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Shift = ElementSizeBits * Op->Index;
|
||||
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
|
||||
"OpSize is too large for VExtractToGPR: {}", OpSize);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
if (SourceSize >= SSERegSize) {
|
||||
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
|
||||
|
||||
const auto GetResult = [&] {
|
||||
if (Shift >= SSEBitSize) {
|
||||
const auto NormalizedShift = Shift - SSEBitSize;
|
||||
return (Src.Upper >> NormalizedShift) & SourceMask;
|
||||
} else {
|
||||
return (Src.Lower >> Shift) & SourceMask;
|
||||
}
|
||||
};
|
||||
|
||||
const auto Result = GetResult();
|
||||
memcpy(GDP, &Result, ElementSize);
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
const uint64_t Result = (Src >> Shift) & SourceMask;
|
||||
GD = Result;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -13,22 +13,46 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
|
||||
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
const auto Scalar = Src2 & Mask;
|
||||
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
|
||||
: Offset;
|
||||
|
||||
// Now shift into place and set all bits but
|
||||
// the ones where we're going to insert our value.
|
||||
Mask <<= ScaledOffset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
Dst |= Src2 << Offset;
|
||||
|
||||
const auto Dst = [&] {
|
||||
if (InUpperLane) {
|
||||
return InterpVector256{
|
||||
.Lower = Src1.Lower,
|
||||
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
|
||||
};
|
||||
} else {
|
||||
return InterpVector256{
|
||||
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
|
||||
.Upper = Src1.Upper,
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
@@ -89,63 +113,73 @@ DEF_OP(Vector_SToF) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
@@ -161,23 +195,26 @@ DEF_OP(Vector_FToF) {
|
||||
// Sometimes is used to convert from a 128bit vector register
|
||||
// in to a 64bit vector register with different sized elements
|
||||
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
|
||||
uint8_t Elements = (OpSize << 1) / Op->SrcElementSize;
|
||||
uint8_t Elements = OpSize == 8 ? 2 : OpSize / Op->SrcElementSize;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
@@ -186,31 +223,31 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
|
||||
+1
-1
@@ -146,7 +146,7 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
|
||||
@@ -154,8 +154,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
@@ -181,13 +179,10 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(SPLATVECTOR2, SplatVector);
|
||||
REGISTER_OP(SPLATVECTOR4, SplatVector);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
@@ -246,18 +241,13 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VINSSCALARELEMENT, VInsScalarElement);
|
||||
REGISTER_OP(VEXTRACTELEMENT, VExtractElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VSLI, VSLI);
|
||||
REGISTER_OP(VSRI, VSRI);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VBITCAST, VBitcast);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
|
||||
@@ -49,7 +49,7 @@ namespace FEXCore::CPU {
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
@@ -181,8 +181,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -207,7 +205,6 @@ namespace FEXCore::CPU {
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -264,18 +261,13 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
@@ -25,93 +25,139 @@ static inline void CacheLineFlush(char *Addr) {
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (IROp->Size) {
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
@@ -144,8 +190,8 @@ DEF_OP(StoreFlag) {
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -158,7 +204,8 @@ DEF_OP(LoadMem) {
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
memset(GDP, 0, 16);
|
||||
|
||||
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
|
||||
@@ -180,16 +227,15 @@ DEF_OP(LoadMem) {
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -221,41 +267,11 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
STORE_DATA(1, uint8_t)
|
||||
STORE_DATA(2, uint16_t)
|
||||
STORE_DATA(4, uint32_t)
|
||||
STORE_DATA(8, uint64_t)
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
|
||||
}
|
||||
#undef STORE_DATA
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
|
||||
@@ -19,13 +19,6 @@ $end_info$
|
||||
#include <sys/random.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(Fence) {
|
||||
|
||||
@@ -30,13 +30,6 @@ DEF_OP(CreateElementPair) {
|
||||
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+858
-598
File diff suppressed because it is too large.
Load diff
+666
-672
File diff suppressed because it is too large.
Load diff
@@ -9,7 +9,7 @@ $end_info$
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
@@ -22,18 +22,18 @@ uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiter
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
@@ -41,7 +41,7 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.Lit = Pointer,
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
@@ -60,20 +60,21 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Bind(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
@@ -87,11 +88,10 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
dc64(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
@@ -103,8 +103,8 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
@@ -117,8 +117,8 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
|
||||
+290
-771
File diff suppressed because it is too large.
Load diff
+157
-187
@@ -6,6 +6,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -17,9 +18,7 @@ $end_info$
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -27,12 +26,11 @@ DEF_OP(SignalReturn) {
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)));
|
||||
br(x0);
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler));
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
|
||||
// spill back to CTX
|
||||
SpillStaticRegs();
|
||||
|
||||
@@ -40,15 +38,15 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
|
||||
ldr(ARMEmitter::WReg::w2, STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter));
|
||||
sub(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 1);
|
||||
str(ARMEmitter::WReg::w2, STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter));
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
add(x2, x2, 8);
|
||||
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 8);
|
||||
str(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
PopCalleeSavedRegisters();
|
||||
|
||||
@@ -59,39 +57,41 @@ DEF_OP(CallbackReturn) {
|
||||
DEF_OP(ExitFunction) {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
Label FullLookup;
|
||||
|
||||
ResetStack();
|
||||
|
||||
aarch64::Register RipReg;
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
ARMEmitter::ForwardLabel l_BranchHost;
|
||||
ARMEmitter::ForwardLabel l_BranchGuest;
|
||||
|
||||
ldr(x0, &l_BranchHost);
|
||||
blr(x0);
|
||||
ldr(ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
|
||||
Bind(&l_BranchHost);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
Bind(&l_BranchGuest);
|
||||
dc64(NewRIP);
|
||||
|
||||
place(&l_BranchHost);
|
||||
place(&l_BranchGuest);
|
||||
} else {
|
||||
RipReg = GetReg<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
ARMEmitter::ForwardLabel FullLookup;
|
||||
auto RipReg = GetReg(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
|
||||
|
||||
ldp(x1, x0, MemOperand(x0));
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
br(x1);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::XReg::x0, RipReg.X());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
br(ARMEmitter::Reg::r1);
|
||||
|
||||
bind(&FullLookup);
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)));
|
||||
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
Bind(&FullLookup);
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
br(TMP1);
|
||||
}
|
||||
}
|
||||
@@ -103,66 +103,65 @@ DEF_OP(Jump) {
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
#define GRFCMP(Node) (Op->CompareSize == 4 ? GetDst(Node).S() : GetDst(Node).D())
|
||||
|
||||
static Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return Condition::eq;
|
||||
case FEXCore::IR::COND_NEQ: return Condition::ne;
|
||||
case FEXCore::IR::COND_SGE: return Condition::ge;
|
||||
case FEXCore::IR::COND_SLT: return Condition::lt;
|
||||
case FEXCore::IR::COND_SGT: return Condition::gt;
|
||||
case FEXCore::IR::COND_SLE: return Condition::le;
|
||||
case FEXCore::IR::COND_UGE: return Condition::cs;
|
||||
case FEXCore::IR::COND_ULT: return Condition::cc;
|
||||
case FEXCore::IR::COND_UGT: return Condition::hi;
|
||||
case FEXCore::IR::COND_ULE: return Condition::ls;
|
||||
case FEXCore::IR::COND_FLU: return Condition::lt;
|
||||
case FEXCore::IR::COND_FGE: return Condition::ge;
|
||||
case FEXCore::IR::COND_FLEU:return Condition::le;
|
||||
case FEXCore::IR::COND_FGT: return Condition::gt;
|
||||
case FEXCore::IR::COND_FU: return Condition::vs;
|
||||
case FEXCore::IR::COND_FNU: return Condition::vc;
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return Condition::nv;
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbnz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (isConst) {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), Const);
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(GRFCMP(Op->Cmp1.ID()), GRFCMP(Op->Cmp2.ID()));
|
||||
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
|
||||
}
|
||||
|
||||
b(TrueTargetLabel, MapBranchCC(Op->Cond));
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
@@ -176,7 +175,7 @@ DEF_OP(Syscall) {
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
SpillStaticRegs();
|
||||
@@ -187,19 +186,25 @@ DEF_OP(Syscall) {
|
||||
}
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(sp, sp, SPOffset);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) continue;
|
||||
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
|
||||
str(GetReg(Op->Header.Args[i].ID()).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
blr(x3);
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
add(sp, sp, SPOffset);
|
||||
// SP supporting move
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY &&
|
||||
(Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
@@ -215,7 +220,7 @@ DEF_OP(Syscall) {
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Move result to its destination register
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -232,30 +237,25 @@ DEF_OP(InlineSyscall) {
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<vixl::aarch64::Register, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
|
||||
x0, x1, x2, x3, x4, x5
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
|
||||
ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5
|
||||
}};
|
||||
|
||||
bool Intersects{};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
if (Reg.GetCode() == x8.GetCode() ||
|
||||
Reg.GetCode() == x4.GetCode() ||
|
||||
Reg.GetCode() == x5.GetCode()) {
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
if (Reg.Idx() == ARMEmitter::Reg::r8.Idx() ||
|
||||
Reg.Idx() == ARMEmitter::Reg::r4.Idx() ||
|
||||
Reg.Idx() == ARMEmitter::Reg::r5.Idx()) {
|
||||
|
||||
SpillMask |= (1U << Reg.GetCode());
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
// XXX: For some reason spilling only the x4, x5, and x8 registers was causing issues
|
||||
// Come back to this once investigation reveals why it fails the gvisor ioctl test
|
||||
// For now override to all GPRs
|
||||
SpillMask = ~0U;
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
@@ -267,66 +267,31 @@ DEF_OP(InlineSyscall) {
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.GetCode() == x8.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == x4.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == x5.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.Idx() == FEXCore::ARMEmitter::Reg::r8.Idx()) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI]));
|
||||
}
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.GetCode() == x8.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == x4.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == x5.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
mov(RegArgs[i], Reg);
|
||||
}
|
||||
else if (Reg.Idx() == FEXCore::ARMEmitter::Reg::r4.Idx()) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
|
||||
}
|
||||
else if (Reg.Idx() == FEXCore::ARMEmitter::Reg::r5.Idx()) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX]));
|
||||
}
|
||||
else {
|
||||
auto Reg = GetReg<RA_32>(Op->Header.Args[i].ID());
|
||||
if (Reg.GetCode() == w8.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == w4.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == w5.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i].W(), Reg);
|
||||
}
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -334,16 +299,11 @@ DEF_OP(InlineSyscall) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i].ID()));
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(x8, Op->HostSyscallNumber);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
@@ -354,16 +314,11 @@ DEF_OP(InlineSyscall) {
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
else {
|
||||
uxtw(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
mov(EmitSize, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -375,13 +330,17 @@ DEF_OP(Thunk) {
|
||||
|
||||
SpillStaticRegs(); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
mov(x0, GetReg<RA_64>(Op->ArgPtr.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
blr(x2);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -394,43 +353,45 @@ DEF_OP(ValidateCode) {
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
LoadConstant(GetReg<RA_64>(Node), 0);
|
||||
LoadConstant(x0, Entry + Op->Offset);
|
||||
LoadConstant(x1, 1);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry + Op->Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 1);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
while (len >= 8)
|
||||
{
|
||||
ldr(x2, MemOperand(x0, idx));
|
||||
LoadConstant(x3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(x2, x3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 8;
|
||||
idx += 8;
|
||||
}
|
||||
while (len >= 4)
|
||||
{
|
||||
ldr(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
ldr(ARMEmitter::WReg::w2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 4;
|
||||
idx += 4;
|
||||
}
|
||||
while (len >= 2)
|
||||
{
|
||||
ldrh(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(const uint16_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
ldrh(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint16_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 2;
|
||||
idx += 2;
|
||||
}
|
||||
while (len >= 1)
|
||||
{
|
||||
ldrb(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(const uint8_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
ldrb(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint8_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 1;
|
||||
idx += 1;
|
||||
}
|
||||
@@ -441,14 +402,18 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Entry);
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
SpillStaticRegs();
|
||||
blr(x2);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
@@ -458,26 +423,31 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, GetReg(Op->Leaf.ID()));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
mov(Dst.first, x0);
|
||||
mov(Dst.second, x1);
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+272
-117
@@ -4,91 +4,157 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
// This is going to be a little gross. Pls forgive me.
|
||||
// Since SVE has the whole vector length agnostic programming
|
||||
// thing going on, we can't exactly freely insert entries into
|
||||
// arbitrary locations in the vector.
|
||||
//
|
||||
// SVE *does* have INSR, however this only shifts the entire
|
||||
// vector to the left by an element size and inserts a value
|
||||
// at the beginning of the vector. Not *quite* what we need.
|
||||
// (though INSR *is* very useful for other things).
|
||||
//
|
||||
// The idea is (in the case of the upper lane), move the upper
|
||||
// lane down, insert into it and recombine with the lower lane.
|
||||
//
|
||||
// In the case of the lower lane, insert and then recombine with
|
||||
// the upper lane.
|
||||
|
||||
if (InUpperLane) {
|
||||
// Move the upper lane down for the insertion.
|
||||
const auto CompactPred = ARMEmitter::PReg::p0;
|
||||
not_(CompactPred, PRED_TMP_32B.Zeroing(), PRED_TMP_16B);
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, DestVector);
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
|
||||
// Put data in place for destructive SPLICE below.
|
||||
mov(Dst.Z(), DestVector.Z());
|
||||
|
||||
// Inserts the GPR value into the given V register.
|
||||
// Also automatically adjusts the index in the case of using the
|
||||
// moved upper lane.
|
||||
const auto Insert = [&](const FEXCore::ARMEmitter::VRegister& reg, int index) {
|
||||
if (InUpperLane) {
|
||||
index -= ElementsPer128Bit;
|
||||
}
|
||||
ins(SubEmitSize, reg, index, Src);
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
Insert(VTMP1, DestIdx);
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), DestVector.Z());
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
} else {
|
||||
mov(Dst.Q(), DestVector.Q());
|
||||
ins(SubEmitSize, Dst, DestIdx, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 4:
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()).W());
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src);
|
||||
break;
|
||||
case 8:
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()).X());
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Src.ID()));
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Src.ID()));
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetVReg(Op->Scalar.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Scalar.ID()).S());
|
||||
fcvt(Dst.D(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Scalar.ID()).D());
|
||||
fcvt(Dst.S(), Src.D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
@@ -96,116 +162,205 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
scvtf(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
} else {
|
||||
scvtf(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
fcvtzs(Dst, SubEmitSize, Mask.Merging(), Vector, SubEmitSize);
|
||||
} else {
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
frinti(SubEmitSize, Dst, Mask.Merging(), Vector);
|
||||
fcvtzs(Dst, SubEmitSize, Mask.Merging(), Dst, SubEmitSize);
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
frinti(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Dst.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// Curiously, FCVTLT and FCVTNT have no bottom variants,
|
||||
// and also interesting is that FCVTLT will iterate the
|
||||
// source vector by accessing each odd element and storing
|
||||
// them consecutively in the destination.
|
||||
//
|
||||
// FCVTNT is somewhat like the opposite. It will read each
|
||||
// consecutive element, but store each result into every odd
|
||||
// element in the destination vector.
|
||||
//
|
||||
// We need to undo the behavior of FCVTNT with UZP2. In the case
|
||||
// of FCVTLT, we instead need to set the vector up with ZIP1, so
|
||||
// that the elements will be processed correctly.
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst, Mask, Vector);
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst, Mask, Vector);
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
} else {
|
||||
switch (Conv) {
|
||||
case 0x0402: // Float <- Half
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(SubEmitSize, Dst.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
case 0x0204: // Half <- Float
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(SubEmitSize, Dst.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
}
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
frinti(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
frintm(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
frintp(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
frintz(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
frinti(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,99 +4,104 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Vector.ID()).V16B());
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesmc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aese(VTMP1, VTMP2);
|
||||
aesmc(VTMP1, VTMP1);
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aese(VTMP1, VTMP2);
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesimc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aesd(VTMP1, VTMP2);
|
||||
aesimc(VTMP1, VTMP1);
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aesd(VTMP1, VTMP2);
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label PastConstant;
|
||||
ARMEmitter::ForwardLabel Constant;
|
||||
ARMEmitter::ForwardLabel PastConstant;
|
||||
|
||||
// Do a "regular" AESE step
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Src.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->Src.ID()).Q());
|
||||
aese(VTMP1, VTMP2);
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
ldr(VTMP3, &ConstantLiteral);
|
||||
ldr(VTMP3.Q(), &Constant);
|
||||
|
||||
// Now EOR in the RCON
|
||||
if (Op->RCON) {
|
||||
tbl(VTMP1.V16B(), VTMP1.V16B(), VTMP3.V16B());
|
||||
tbl(VTMP1.Q(), VTMP1.Q(), VTMP3.Q());
|
||||
|
||||
LoadConstant(TMP1.W(), Op->RCON);
|
||||
ins(VTMP2.V4S(), 1, TMP1.W());
|
||||
ins(VTMP2.V4S(), 3, TMP1.W());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), VTMP2.V16B());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32);
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP2.Q(), TMP1);
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), VTMP2.Q());
|
||||
}
|
||||
else {
|
||||
tbl(GetDst(Node).V16B(), VTMP1.V16B(), VTMP3.V16B());
|
||||
tbl(GetVReg(Node).Q(), VTMP1.Q(), VTMP3.Q());
|
||||
}
|
||||
|
||||
b(&PastConstant);
|
||||
place(&ConstantLiteral);
|
||||
bind(&PastConstant);
|
||||
Bind(&Constant);
|
||||
dc64(0x040B0E01'0B0E0104ULL);
|
||||
dc64(0x0C030609'0306090CULL);
|
||||
Bind(&PastConstant);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32cb(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
crc32cb(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 2:
|
||||
crc32ch(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
crc32ch(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 4:
|
||||
crc32cw(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
crc32cw(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_64>(Op->Src2.ID()));
|
||||
crc32cx(Dst, Src1, Src2);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
@@ -105,24 +110,24 @@ DEF_OP(CRC32) {
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node).Q();
|
||||
auto Src1 = GetSrc(Op->Src1.ID()).V2D();
|
||||
auto Src2 = GetSrc(Op->Src2.ID()).V2D();
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src1 = GetVReg(Op->Src1.ID());
|
||||
auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
pmull(Dst, Src1, Src2);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
|
||||
break;
|
||||
case 0b00000001:
|
||||
mov(VTMP1.V1D(), Src1, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src2);
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
|
||||
break;
|
||||
case 0b00010000:
|
||||
mov(VTMP1.V1D(), Src2, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src1);
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
|
||||
break;
|
||||
case 0b00010001:
|
||||
pmull2(Dst, Src1, Src2);
|
||||
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
|
||||
@@ -7,13 +7,10 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Value.ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+242
-226
@@ -11,6 +11,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
@@ -28,6 +29,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -76,10 +78,7 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -90,11 +89,16 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
uxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, uint16_t>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -105,38 +109,49 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_F80_F32:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, float>(ARMEmitter::Reg::r0);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, double>(ARMEmitter::Reg::r0);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -144,221 +159,296 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
uxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
}
|
||||
else {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
}
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint32_t>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<float, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
fmov(GetDst(Node).S(), v0.S());
|
||||
const auto Dst = GetVReg(Node);
|
||||
fmov(Dst.S(), ARMEmitter::SReg::s0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), ARMEmitter::DReg::d0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double>(ARMEmitter::Reg::r0);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), ARMEmitter::DReg::d0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
mov(ARMEmitter::DReg::d1, Src2.D());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double, double>(ARMEmitter::Reg::r0);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), ARMEmitter::DReg::d0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
uxth(GetReg<RA_64>(Node), x0);
|
||||
const auto Dst = GetReg(Node);
|
||||
uxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetReg<RA_32>(Node), w0);
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r3, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
#endif
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r3, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -378,7 +468,6 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
@@ -387,25 +476,22 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
if (vixl::IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
emit.blr(FEXCore::ARMEmitter::Reg::r0);
|
||||
emit.Bind(&l_BranchHost);
|
||||
emit.dc64(LinkerAddress);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
});
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
@@ -420,7 +506,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
@@ -431,10 +517,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
#if DEBUG
|
||||
Decoder.AppendVisitor(&Disasm)
|
||||
#endif
|
||||
|
||||
uint32_t NumUsedGPRs = NumGPRs;
|
||||
uint32_t NumUsedGPRPairs = NumGPRPairs;
|
||||
uint32_t UsedRegisterCount = RegisterCount;
|
||||
@@ -473,7 +555,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -494,7 +576,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
@@ -502,7 +584,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
SetAllowAssembler(true);
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
@@ -511,6 +592,7 @@ void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
@@ -519,20 +601,20 @@ void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::Arm64JITCore::";
|
||||
auto Buffer = GetBuffer();
|
||||
Buffer->EmitString(JITString);
|
||||
Buffer->Align();
|
||||
EmitString(JITString);
|
||||
Align();
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
@@ -540,84 +622,6 @@ Arm64JITCore::~Arm64JITCore() {
|
||||
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg].W();
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg].W();
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
return RA32Pair[Reg];
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
return RA64Pair[Reg];
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
@@ -655,7 +659,6 @@ FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
|
||||
}
|
||||
|
||||
|
||||
bool Arm64JITCore::IsFPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
@@ -673,7 +676,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
using namespace aarch64;
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
@@ -681,9 +685,13 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x0, Entry);
|
||||
#endif
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
#endif
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry);
|
||||
#endif
|
||||
|
||||
this->IR = IR;
|
||||
|
||||
@@ -717,7 +725,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
GetBuffer()->CursorForward(GDBSize);
|
||||
CursorIncrement(GDBSize);
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -725,11 +733,13 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
sub(sp, sp, SpillSlots * 16);
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (vixl::aarch64::Assembler::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
sub(sp, sp, x0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, TotalSpillSlotsSize);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, ARMEmitter::XReg::x0, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -754,7 +764,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
bind(&IsTarget->second);
|
||||
Bind(&IsTarget->second);
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
@@ -780,10 +790,13 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = GetCursorAddress<uint8_t *>();
|
||||
CPU.EnsureIAndDCacheCoherency(GuestEntry, CodeEnd - GuestEntry);
|
||||
ClearICache(GuestEntry, CodeEnd - GuestEntry);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
#endif
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = CodeEnd - GuestEntry;
|
||||
@@ -796,15 +809,18 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
if (SpillSlots == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (vixl::aarch64::Assembler::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, TotalSpillSlotsSize);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, ARMEmitter::XReg::x0, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+60
-47
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
@@ -23,24 +24,11 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
#define TMP3 x2
|
||||
#define TMP4 x3
|
||||
|
||||
#define VTMP1 v1
|
||||
#define VTMP2 v2
|
||||
#define VTMP3 v3
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
@@ -68,12 +56,12 @@ private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
const bool HostSupportsSVE{};
|
||||
|
||||
Label *PendingTargetLabel;
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
|
||||
std::map<IR::NodeID, aarch64::Label> JumpTargets;
|
||||
std::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
@@ -96,38 +84,72 @@ private:
|
||||
constexpr static uint8_t RA_64 = 1;
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] aarch64::Register GetReg(IR::NodeID Node) const;
|
||||
[[nodiscard]] FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
template<>
|
||||
[[nodiscard]] aarch64::Register GetReg<RA_32>(IR::NodeID Node) const;
|
||||
template<>
|
||||
[[nodiscard]] aarch64::Register GetReg<RA_64>(IR::NodeID Node) const;
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair(IR::NodeID Node) const;
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
}
|
||||
|
||||
template<>
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(IR::NodeID Node) const;
|
||||
template<>
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(IR::NodeID Node) const;
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]] aarch64::VRegister GetSrc(IR::NodeID Node) const;
|
||||
[[nodiscard]] aarch64::VRegister GetDst(IR::NodeID Node) const;
|
||||
[[nodiscard]] FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]] std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return RA64Pair[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const;
|
||||
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] MemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
aarch64::Register Base,
|
||||
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]] FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
@@ -161,7 +183,8 @@ private:
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
ARMEmitter::ForwardLabel Loc;
|
||||
uint64_t Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
@@ -171,7 +194,7 @@ private:
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
@@ -179,7 +202,7 @@ private:
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
@@ -212,7 +235,7 @@ private:
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
@@ -224,7 +247,7 @@ private:
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
@@ -344,8 +367,6 @@ private:
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -365,13 +386,10 @@ private:
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector2);
|
||||
DEF_OP(SplatVector4);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -430,18 +448,13 @@ private:
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
+847
-444
File diff suppressed because it is too large.
Load diff
+81
-77
@@ -5,13 +5,12 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <syscall.h>
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
@@ -23,13 +22,13 @@ DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val:
|
||||
dmb(FullSystem, BarrierReads);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::LD);
|
||||
break;
|
||||
case IR::Fence_LoadStore.Val:
|
||||
dmb(FullSystem, BarrierAll);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::SY);
|
||||
break;
|
||||
case IR::Fence_Store.Val:
|
||||
dmb(FullSystem, BarrierWrites);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ST);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
@@ -41,114 +40,120 @@ DEF_OP(Break) {
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Op->Reason.Signal,
|
||||
.TrapNo = Op->Reason.TrapNumber,
|
||||
.si_code = Op->Reason.si_code,
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
|
||||
br(TMP1);
|
||||
break;
|
||||
default:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
mrs(Dst, FPCR);
|
||||
lsr(Dst, Dst, 22);
|
||||
auto Dst = GetReg(Node);
|
||||
mrs(Dst, ARMEmitter::SystemRegister::FPCR);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst, Dst, 22);
|
||||
|
||||
// FTZ is already in the correct location
|
||||
// Rounding mode is different
|
||||
and_(TMP1, Dst, 0b11);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dst, 0b11);
|
||||
|
||||
cmp(TMP1, 1);
|
||||
LoadConstant(TMP3, IR::ROUND_MODE_POSITIVE_INFINITY);
|
||||
csel(TMP2, TMP3, xzr, vixl::aarch64::Condition::eq);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, 1);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, IR::ROUND_MODE_POSITIVE_INFINITY);
|
||||
csel(ARMEmitter::Size::i64Bit, TMP2, TMP3, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
cmp(TMP1, 2);
|
||||
LoadConstant(TMP3, IR::ROUND_MODE_NEGATIVE_INFINITY);
|
||||
csel(TMP2, TMP3, TMP2, vixl::aarch64::Condition::eq);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, 2);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, IR::ROUND_MODE_NEGATIVE_INFINITY);
|
||||
csel(ARMEmitter::Size::i64Bit, TMP2, TMP3, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
cmp(TMP1, 3);
|
||||
LoadConstant(TMP3, IR::ROUND_MODE_TOWARDS_ZERO);
|
||||
csel(TMP2, TMP3, TMP2, vixl::aarch64::Condition::eq);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, 3);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, IR::ROUND_MODE_TOWARDS_ZERO);
|
||||
csel(ARMEmitter::Size::i64Bit, TMP2, TMP3, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
orr(Dst, Dst, TMP2);
|
||||
orr(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2.R());
|
||||
|
||||
bfi(Dst, TMP2, 0, 2);
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, 0, 2);
|
||||
}
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg<RA_64>(Op->RoundMode.ID());
|
||||
auto Src = GetReg(Op->RoundMode.ID());
|
||||
|
||||
// Setup the rounding flags correctly
|
||||
and_(TMP1, Src, 0b11);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Src, 0b11);
|
||||
|
||||
cmp(TMP1, IR::ROUND_MODE_POSITIVE_INFINITY);
|
||||
LoadConstant(TMP3, 1);
|
||||
csel(TMP2, TMP3, xzr, vixl::aarch64::Condition::eq);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, IR::ROUND_MODE_POSITIVE_INFINITY);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, 1);
|
||||
csel(ARMEmitter::Size::i64Bit, TMP2, TMP3, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
cmp(TMP1, IR::ROUND_MODE_NEGATIVE_INFINITY);
|
||||
LoadConstant(TMP3, 2);
|
||||
csel(TMP2, TMP3, TMP2, vixl::aarch64::Condition::eq);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, IR::ROUND_MODE_NEGATIVE_INFINITY);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, 2);
|
||||
csel(ARMEmitter::Size::i64Bit, TMP2, TMP3, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
cmp(TMP1, IR::ROUND_MODE_TOWARDS_ZERO);
|
||||
LoadConstant(TMP3, 3);
|
||||
csel(TMP2, TMP3, TMP2, vixl::aarch64::Condition::eq);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, IR::ROUND_MODE_TOWARDS_ZERO);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, 3);
|
||||
csel(ARMEmitter::Size::i64Bit, TMP2, TMP3, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
mrs(TMP1, FPCR);
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// vixl simulator doesn't support anything beyond ties-to-even rounding
|
||||
#ifndef VIXL_SIMULATOR
|
||||
// Insert the rounding flags
|
||||
bfi(TMP1, TMP2, 22, 2);
|
||||
bfi(ARMEmitter::Size::i64Bit, TMP1, TMP2, 22, 2);
|
||||
#endif
|
||||
|
||||
// Insert the FTZ flag
|
||||
lsr(TMP2, Src, 2);
|
||||
bfi(TMP1, TMP2, 24, 1);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, Src, 2);
|
||||
bfi(ARMEmitter::Size::i64Bit, TMP1, TMP2, 24, 1);
|
||||
|
||||
// Now save the new FPCR
|
||||
msr(FPCR, TMP1);
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
|
||||
}
|
||||
else {
|
||||
fmov(x0, GetSrc(Op->Value.ID()).V1D());
|
||||
// Bug in vixl that source vector needs to b V1D rather than V2D?
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value.ID()), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value.ID()), true);
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
@@ -166,27 +171,27 @@ DEF_OP(ProcessorID) {
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Allocate some temporary space for storing the uint32_t CPU and Node IDs
|
||||
sub(sp, sp, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Load the getcpu syscall number
|
||||
LoadConstant(x8, SYS_getcpu);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_getcpu);
|
||||
|
||||
// CPU pointer in x0
|
||||
add(x0, sp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
|
||||
// Node in x1
|
||||
add(x1, sp, 4);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 4);
|
||||
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
// Load the values returned by the kernel
|
||||
ldp(w0, w1, MemOperand(sp));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::WReg::w0, ARMEmitter::WReg::w1, ARMEmitter::Reg::rsp);
|
||||
// Deallocate stack space
|
||||
sub(sp, sp, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
@@ -194,14 +199,13 @@ DEF_OP(ProcessorID) {
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now store the result in the destination in the expected format
|
||||
// uint32_t Res = (node << 12) | cpu;
|
||||
// CPU is in w0
|
||||
// Node is in w1
|
||||
orr(GetReg<RA_64>(Node), x0, Operand(x1, LSL, 12));
|
||||
orr(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ShiftType::LSL, 12);
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
@@ -209,21 +213,21 @@ DEF_OP(RDRAND) {
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
auto Dst = GetRegPair(Node);
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
mrs(Dst.first, RNDRRS);
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDRRS);
|
||||
}
|
||||
else {
|
||||
mrs(Dst.first, RNDR);
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDR);
|
||||
}
|
||||
|
||||
// If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100
|
||||
cset(Dst.second, Condition::ne);
|
||||
cset(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Condition::CC_NE);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
hint(SystemHint::YIELD);
|
||||
yield();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+22
-55
@@ -7,78 +7,45 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(Op->Header.Size == 4 || Op->Header.Size == 8, "Invalid size");
|
||||
const auto EmitSize = Op->Header.Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Src = GetRegPair(Op->Pair.ID());
|
||||
const std::array<ARMEmitter::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov(EmitSize, GetReg(Node), Regs[Op->Element]);
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
std::pair<aarch64::Register, aarch64::Register> Dst;
|
||||
aarch64::Register RegFirst;
|
||||
aarch64::Register RegSecond;
|
||||
aarch64::Register RegTmp;
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Invalid size");
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> Dst = GetRegPair(Node);
|
||||
ARMEmitter::Register RegFirst = GetReg(Op->Lower.ID());
|
||||
ARMEmitter::Register RegSecond = GetReg(Op->Upper.ID());
|
||||
ARMEmitter::Register RegTmp = TMP1.R();
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetReg<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Upper.ID());
|
||||
RegTmp = w0;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetReg<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Upper.ID());
|
||||
RegTmp = x0;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
|
||||
}
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (Dst.first.GetCode() != RegSecond.GetCode()) {
|
||||
mov(Dst.first, RegFirst);
|
||||
mov(Dst.second, RegSecond);
|
||||
} else if (Dst.second.GetCode() != RegFirst.GetCode()) {
|
||||
mov(Dst.second, RegSecond);
|
||||
mov(Dst.first, RegFirst);
|
||||
if (Dst.first.Idx() != RegSecond.Idx()) {
|
||||
mov(EmitSize, Dst.first, RegFirst);
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
} else if (Dst.second.Idx() != RegFirst.Idx()) {
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
mov(EmitSize, Dst.first, RegFirst);
|
||||
} else {
|
||||
mov(RegTmp, RegFirst);
|
||||
mov(Dst.second, RegSecond);
|
||||
mov(Dst.first, RegTmp);
|
||||
mov(EmitSize, RegTmp, RegFirst);
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
mov(EmitSize, Dst.first, RegTmp);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+2429
-1949
File diff suppressed because it is too large.
Load diff
+44
-11
@@ -1143,26 +1143,59 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
|
||||
const auto Is256Bit = Offset >= SSEBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrb(GetDst<RA_32>(Node), xmm15, Op->Index - 16);
|
||||
} else {
|
||||
pextrb(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrw(GetDst<RA_32>(Node), xmm15, Op->Index - 8);
|
||||
} else {
|
||||
pextrw(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrd(GetDst<RA_32>(Node), xmm15, Op->Index - 4);
|
||||
} else {
|
||||
pextrd(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrq(GetDst<RA_64>(Node), xmm15, Op->Index - 2);
|
||||
} else {
|
||||
pextrq(GetDst<RA_64>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
@@ -42,7 +42,7 @@ DEF_OP(SignalReturn) {
|
||||
DEF_OP(CallbackReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
@@ -71,7 +71,7 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
uint64_t NewRIP;
|
||||
|
||||
+226
-74
@@ -17,27 +17,76 @@ namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
if (InUpperLane && !Is256Bit) {
|
||||
LOGMAN_MSG_A_FMT("Attempt to access upper 128-bit lane in 128-bit operation! Offset={}",
|
||||
Offset);
|
||||
return;
|
||||
}
|
||||
|
||||
if (Is256Bit) {
|
||||
vmovapd(ToYMM(Dst), ToYMM(DestVector));
|
||||
} else {
|
||||
vmovapd(Dst, DestVector);
|
||||
}
|
||||
|
||||
const auto Insert = [&](const Xbyak::Xmm& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
pinsrb(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
pinsrw(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
pinsrd(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
pinsrq(reg, GetSrc<RA_64>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
vextracti128(xmm15, ToYMM(Dst), 1);
|
||||
Insert(xmm15, DestIdx);
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,8 +112,10 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
@@ -83,6 +134,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -104,99 +159,196 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtdq2ps(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtdq2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
// There is no vector form of this instruction until AVX512VL + AVX512DQ (vcvtqq2pd)
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
pextrq(rax, Vector, 1);
|
||||
pextrq(rcx, Vector, 0);
|
||||
cvtsi2sd(Dst, rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
movlhps(Dst, xmm15);
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
|
||||
pextrq(rax, xmm15, 1);
|
||||
pextrq(rcx, xmm15, 0);
|
||||
cvtsi2sd(xmm15, rcx);
|
||||
cvtsi2sd(xmm14, rax);
|
||||
movlhps(xmm15, xmm14);
|
||||
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
} else {
|
||||
vmovlhps(Dst, Dst, xmm15);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvttps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvttpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvtpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtps2pd(ToYMM(Dst), Vector);
|
||||
} else {
|
||||
vcvtps2pd(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtpd2ps(Dst, ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t RoundMode{};
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
RoundMode = 0b0000'0'0'00;
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'01;
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'10;
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
RoundMode = 0b0000'0'0'11;
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
RoundMode = 0b0000'0'1'00;
|
||||
break;
|
||||
}
|
||||
const uint8_t RoundMode = [Op] {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
return 0b0000'0'0'00;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
return 0b0000'0'0'01;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
return 0b0000'0'0'10;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
return 0b0000'0'0'11;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
return 0b0000'0'1'00;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled rounding mode");
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundps(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundps(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundpd(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundpd(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+30
-17
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -60,32 +61,42 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
const auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
sub(rsp, AVXRegSize * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
movaps(ptr[rsp + i * 16], RAXMM_x[i]);
|
||||
vmovups(ptr[rsp + i * AVXRegSize], ToYMM(RAXMM_x[i]));
|
||||
}
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
for (const auto &Reg : RA64) {
|
||||
push(Reg);
|
||||
}
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
const auto NumPush = RA64.size();
|
||||
if ((NumPush & 1) != 0) {
|
||||
// Align
|
||||
sub(rsp, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void X86JITCore::PopRegs() {
|
||||
auto NumPush = RA64.size();
|
||||
const auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
movaps(RAXMM_x[i], ptr[rsp + i * 16]);
|
||||
if ((NumPush & 1) != 0) {
|
||||
// Align
|
||||
add(rsp, 8);
|
||||
}
|
||||
|
||||
add(rsp, 16 * RAXMM_x.size());
|
||||
for (uint32_t i = RA64.size(); i > 0; --i) {
|
||||
pop(RA64[i - 1]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
vmovups(ToYMM(RAXMM_x[i]), ptr[rsp + i * AVXRegSize]);
|
||||
}
|
||||
|
||||
add(rsp, AVXRegSize * RAXMM_x.size());
|
||||
}
|
||||
|
||||
void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
@@ -360,7 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
|
||||
{
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -572,6 +583,8 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("x86::CompileCode");
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
@@ -599,7 +612,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
sub(rsp, SpillSlots * 16);
|
||||
sub(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
|
||||
@@ -209,7 +209,7 @@ private:
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
using JCC = void (X86JITCore::*)(const Label& label, LabelType type);
|
||||
@@ -337,6 +337,8 @@ private:
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
@@ -345,8 +347,6 @@ private:
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -366,12 +366,10 @@ private:
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -430,18 +428,13 @@ private:
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
+365
-189
@@ -21,130 +21,266 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(GetDst<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(GetDst<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(GetDst<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid GPR load of size 16");
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(GetDst(Node), dword [STATE + Op->Offset]);
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(GetDst(Node), qword [STATE + Op->Offset]);
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16:
|
||||
LogMan::Msg::DFmt("Invalid store size of 16");
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid store size of 16");
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrb(byte [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrw(word [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovd(dword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovq(qword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetSrc<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
movzx(GetSrc<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
mov(GetSrc<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
mov(GetSrc<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetSrc(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -153,21 +289,21 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 2:
|
||||
movzx(GetDst<RA_32>(Node), word [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), word [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 4:
|
||||
mov(GetDst<RA_32>(Node), dword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_32>(Node), dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_64>(Node), qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -186,53 +322,63 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(eax, byte [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, byte [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(eax, word [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, word [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [rax + index * Op->Stride]);
|
||||
vmovd(Dst, dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
|
||||
vmovq(Dst, qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pinsrb(GetDst(Node), byte [STATE + rax], 0);
|
||||
pinsrb(Dst, byte [STATE + rax], 0);
|
||||
break;
|
||||
case 2:
|
||||
pinsrw(GetDst(Node), word [STATE + rax], 0);
|
||||
pinsrw(Dst, word [STATE + rax], 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [STATE + rax]);
|
||||
vmovd(Dst, dword [STATE + rax]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [STATE + rax]);
|
||||
vmovq(Dst, qword [STATE + rax]);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + rax]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + rax]);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + rax]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + rax]);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
vmovups(ToYMM(Dst), yword [STATE + rax]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -245,12 +391,13 @@ DEF_OP(LoadContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
size_t size = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto value = GetSrc<RA_64>(Op->Value.ID());
|
||||
const auto Value = GetSrc<RA_64>(Op->Value.ID());
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -258,10 +405,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
if (!(OpSize == 1 || OpSize == 2 || OpSize == 4 || OpSize == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
}
|
||||
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
mov(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -270,57 +417,64 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(xword [STATE + rax], value);
|
||||
else
|
||||
movups(xword [STATE + rax], value);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(xword [STATE + rax], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + rax], Value);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [STATE + rax], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -333,10 +487,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
@@ -355,36 +509,44 @@ DEF_OP(SpillRegister) {
|
||||
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Src = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(dword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movss(dword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(qword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movsd(qword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(xword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movaps(xword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(yword [rsp + SlotOffset], ToYMM(Src));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
DEF_OP(FillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
@@ -403,23 +565,33 @@ DEF_OP(FillRegister) {
|
||||
mov(GetDst<RA_64>(Node), qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(GetDst(Node), dword [rsp + SlotOffset]);
|
||||
vmovss(Dst, dword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(GetDst(Node), qword [rsp + SlotOffset]);
|
||||
vmovsd(Dst, qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(GetDst(Node), xword [rsp + SlotOffset]);
|
||||
vmovaps(Dst, xword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
@@ -464,130 +636,136 @@ Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
const auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx (Dst, byte [MemPtr]);
|
||||
movzx(Dst, byte [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx (Dst, word [MemPtr]);
|
||||
movzx(Dst, word [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(Dst.cvt32(), dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
auto Dst = GetDst(Node);
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(eax, byte [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(eax, word [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(Dst, dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
else
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, GetDst(Node));
|
||||
}
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
vmovups(Dst, xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 2:
|
||||
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 4:
|
||||
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 8:
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrb(byte [MemPtr], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrw(word [MemPtr], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovd(dword [MemPtr], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovq(qword [MemPtr], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
vmovups(xword [MemPtr], Value);
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [MemPtr], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
@@ -618,8 +796,8 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP(LOADREGISTER, Unhandled); // SRA specific, not supported on this backend
|
||||
REGISTER_OP(STOREREGISTER, Unhandled);
|
||||
REGISTER_OP(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
@@ -630,8 +808,6 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
|
||||
@@ -47,14 +47,22 @@ DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)], Op->Reason.Signal);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], Op->Reason.TrapNumber);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], Op->Reason.ErrorRegister);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], Op->Reason.si_code);
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Op->Reason.Signal,
|
||||
.TrapNo = Op->Reason.TrapNumber,
|
||||
.si_code = Op->Reason.si_code,
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
mov(TMP1, Constant);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData)], TMP1);
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
|
||||
@@ -73,17 +73,11 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+3091
-1097
File diff suppressed because it is too large.
Load diff
+20
-14
@@ -17,6 +17,10 @@ namespace FEXCore {
|
||||
LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma.new_object<BlockLinksMapType>();
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
// |
|
||||
@@ -29,46 +33,48 @@ LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
// Allocate a region of memory that we can use to back our block pointers
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, ctx->Config.VirtualMemSize / 4096 * 8, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, TotalCacheSize, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
// We need 32KB per guest page (One pointer per byte)
|
||||
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_AA_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8);
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PageMemory), CODE_SIZE);
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
|
||||
const size_t TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
|
||||
// No need to free BlockLinks map.
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8, MADV_DONTNEED);
|
||||
madvise(reinterpret_cast<void*>(PageMemory), CODE_SIZE, MADV_DONTNEED);
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, MADV_DONTNEED);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1
|
||||
madvise(reinterpret_cast<void*>(L1Pointer), L1_SIZE, MADV_DONTNEED);
|
||||
// Clear L2
|
||||
ClearL2Cache();
|
||||
// All code is gone, remove links
|
||||
BlockLinks.clear();
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
madvise(reinterpret_cast<void*>(PagePointer), TotalCacheSize, MADV_DONTNEED);
|
||||
// Clear the BlockLinks allocator which frees the BlockLinks map implicitly.
|
||||
BlockLinks_mbr.release();
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma.new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
BlockList.clear();
|
||||
}
|
||||
|
||||
+26
-13
@@ -4,10 +4,12 @@
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <memory_resource>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -17,7 +19,7 @@ namespace Context {
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
struct LookupCacheEntry {
|
||||
struct LookupCacheEntry {
|
||||
uintptr_t HostCode;
|
||||
uintptr_t GuestCode;
|
||||
};
|
||||
@@ -54,7 +56,7 @@ public:
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
@@ -62,7 +64,7 @@ public:
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
@@ -73,7 +75,7 @@ public:
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -88,7 +90,7 @@ public:
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
@@ -104,9 +106,9 @@ public:
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks.lower_bound({Address, 0});
|
||||
auto upper = BlockLinks.upper_bound({Address, UINTPTR_MAX});
|
||||
for (auto it = lower; it != upper; it = BlockLinks.erase(it)) {
|
||||
auto lower = BlockLinks->lower_bound({Address, 0});
|
||||
auto upper = BlockLinks->upper_bound({Address, UINTPTR_MAX});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second();
|
||||
}
|
||||
|
||||
@@ -144,7 +146,7 @@ public:
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks.insert({{GuestDestination, HostLink}, delinker});
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
@@ -158,7 +160,7 @@ public:
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
@@ -167,7 +169,7 @@ public:
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
@@ -237,9 +239,20 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
// Use a monotonic buffer resource to allocate both the std::pmr::map and its members.
|
||||
// This allows us to quickly clear the block link map by clearing the monotonic allocator.
|
||||
// If we had allocated the block link map without the MBR, then clearing the map would require slowly
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
|
||||
std::pmr::polymorphic_allocator<std::byte> BlockLinks_pma {&BlockLinks_mbr};
|
||||
BlockLinksMapType *BlockLinks;
|
||||
|
||||
std::map<BlockLinkTag, std::function<void()>> BlockLinks;
|
||||
std::map<uint64_t, uint64_t> BlockList;
|
||||
tsl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
|
||||
+1135
-406
File diff suppressed because it is too large.
Load diff
+209
-25
@@ -76,10 +76,10 @@ public:
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::Context *CTX{};
|
||||
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
@@ -153,8 +153,9 @@ public:
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
|
||||
void ResetWorkingList();
|
||||
void ResetDecodeFailure() { DecodeFailure = false; }
|
||||
void ResetDecodeFailure() { NeedsBlockEnd = DecodeFailure = false; }
|
||||
bool HadDecodeFailure() const { return DecodeFailure; }
|
||||
bool NeedsBlockEnder() const { return NeedsBlockEnd; }
|
||||
|
||||
void BeginFunction(uint64_t RIP, std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void Finalize();
|
||||
@@ -278,6 +279,12 @@ public:
|
||||
void NOTOp(OpcodeArgs);
|
||||
void XADDOp(OpcodeArgs);
|
||||
void PopcountOp(OpcodeArgs);
|
||||
void DAAOp(OpcodeArgs);
|
||||
void DASOp(OpcodeArgs);
|
||||
void AAAOp(OpcodeArgs);
|
||||
void AASOp(OpcodeArgs);
|
||||
void AAMOp(OpcodeArgs);
|
||||
void AADOp(OpcodeArgs);
|
||||
void XLATOp(OpcodeArgs);
|
||||
template<bool Reseed>
|
||||
void RDRANDOp(OpcodeArgs);
|
||||
@@ -292,6 +299,8 @@ public:
|
||||
void WriteSegmentReg(OpcodeArgs);
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
|
||||
// SSE
|
||||
void MOVAPSOp(OpcodeArgs);
|
||||
void MOVUPSOp(OpcodeArgs);
|
||||
@@ -314,10 +323,6 @@ public:
|
||||
|
||||
void MOVQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PADDQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSUBQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void MOVMSKOp(OpcodeArgs);
|
||||
void MOVMSKOpOne(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
@@ -342,8 +347,6 @@ public:
|
||||
void PSLLDQ(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRAIOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PAVGOp(OpcodeArgs);
|
||||
void MOVDDUPOp(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void CVTGPR_To_FPR(OpcodeArgs);
|
||||
@@ -370,15 +373,16 @@ public:
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void SHUFOp(OpcodeArgs);
|
||||
void ANDNOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PINSROp(OpcodeArgs);
|
||||
void InsertPSOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PExtrOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
template <size_t ElementSize>
|
||||
void PSIGN(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void VPSIGN(OpcodeArgs);
|
||||
|
||||
// BMI1 Ops
|
||||
void ANDNBMIOp(OpcodeArgs);
|
||||
@@ -398,6 +402,117 @@ public:
|
||||
// ADX Ops
|
||||
void ADXOp(OpcodeArgs);
|
||||
|
||||
// AVX Ops
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void AVXVectorUnaryOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void AVXExtendVectorElements(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
void AVXVectorRound(OpcodeArgs);
|
||||
|
||||
template <size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void AVXVector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template <size_t SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
void AVXVFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VADDSUBPOp(OpcodeArgs);
|
||||
|
||||
void VAESDecOp(OpcodeArgs);
|
||||
void VAESDecLastOp(OpcodeArgs);
|
||||
void VAESEncOp(OpcodeArgs);
|
||||
void VAESEncLastOp(OpcodeArgs);
|
||||
void VAESIMCOp(OpcodeArgs);
|
||||
void VAESKeyGenAssistOp(OpcodeArgs);
|
||||
|
||||
void VANDNOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VBROADCASTOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VDPPOp(OpcodeArgs);
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void VHADDPOp(OpcodeArgs);
|
||||
|
||||
void VINSERTOp(OpcodeArgs);
|
||||
void VINSERTPSOp(OpcodeArgs);
|
||||
|
||||
void VMOVAPS_VMOVAPD_Op(OpcodeArgs);
|
||||
void VMOVUPS_VMOVUPD_Op(OpcodeArgs);
|
||||
|
||||
void VMOVHPOp(OpcodeArgs);
|
||||
void VMOVLPOp(OpcodeArgs);
|
||||
|
||||
void VMOVDDUPOp(OpcodeArgs);
|
||||
void VMOVSHDUPOp(OpcodeArgs);
|
||||
void VMOVSLDUPOp(OpcodeArgs);
|
||||
|
||||
void VMOVVectorNTOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPACKSSOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPACKUSOp(OpcodeArgs);
|
||||
|
||||
void VPERM2Op(OpcodeArgs);
|
||||
void VPERMQOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPERMILImmOp(OpcodeArgs);
|
||||
|
||||
void VPHMINPOSUWOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPHSUBOp(OpcodeArgs);
|
||||
|
||||
void VPMULHRSWOp(OpcodeArgs);
|
||||
|
||||
template <bool Signed>
|
||||
void VPMULHWOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Signed>
|
||||
void VPMULLOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSLLOp(OpcodeArgs);
|
||||
void VPSLLDQOp(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void VPSLLIOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSRAOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSRAIOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSRLDOp(OpcodeArgs);
|
||||
void VPSRLDQOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPUNPCKHOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPUNPCKLOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSRLIOp(OpcodeArgs);
|
||||
|
||||
void VZEROOp(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
template<size_t width>
|
||||
void FLD(OpcodeArgs);
|
||||
@@ -515,7 +630,7 @@ public:
|
||||
void X87FRSTORF64(OpcodeArgs);
|
||||
void X87FXAMF64(OpcodeArgs);
|
||||
void X87LDENVF64(OpcodeArgs);
|
||||
|
||||
|
||||
template<size_t width, bool Integer, FCOMIFlags whichflags, bool poptwice>
|
||||
void FCOMIF64(OpcodeArgs);
|
||||
|
||||
@@ -540,12 +655,6 @@ public:
|
||||
template<bool ToXMM>
|
||||
void MOVQ2DQ(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void PADDSOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void PSUBSOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void ADDSUBPOp(OpcodeArgs);
|
||||
|
||||
@@ -570,12 +679,7 @@ public:
|
||||
|
||||
void MOVBEOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void HADDP(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void HSUBP(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void PHADD(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PHSUB(OpcodeArgs);
|
||||
|
||||
@@ -632,8 +736,6 @@ public:
|
||||
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
|
||||
OrderedNode *GetPackedRFLAG(bool Lower8);
|
||||
|
||||
@@ -642,10 +744,87 @@ public:
|
||||
bool HandledLock = false;
|
||||
private:
|
||||
bool DecodeFailure{false};
|
||||
bool NeedsBlockEnd{false};
|
||||
FEXCore::IR::IROp_IRHeader *Current_Header{};
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
|
||||
// Opcode helpers for generalizing behavior across VEX and non-VEX variants.
|
||||
|
||||
OrderedNode* ADDSUBPOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
|
||||
OrderedNode* AESKeyGenAssistImpl(OpcodeArgs);
|
||||
OrderedNode* AESIMCImpl(OpcodeArgs);
|
||||
|
||||
OrderedNode* DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
|
||||
OrderedNode* ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize,
|
||||
size_t DstElementSize, bool Signed);
|
||||
|
||||
OrderedNode* InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* PACKSSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PACKUSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PHMINPOSUWOpImpl(OpcodeArgs);
|
||||
|
||||
OrderedNode* PHSUBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2, size_t ElementSize);
|
||||
|
||||
OrderedNode* PMULHRSWOpImpl(OpcodeArgs, OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PMULHWOpImpl(OpcodeArgs, bool Signed,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PMULLOpImpl(OpcodeArgs, size_t ElementSize, bool Signed,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PSIGNImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PSLLIImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Shift);
|
||||
|
||||
OrderedNode* PSLLImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, OrderedNode *ShiftVec);
|
||||
|
||||
OrderedNode* PSRAOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, OrderedNode *ShiftVec);
|
||||
|
||||
OrderedNode* PSRLDOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, OrderedNode *ShiftVec);
|
||||
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
|
||||
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType);
|
||||
|
||||
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Mode);
|
||||
|
||||
OrderedNode* Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
@@ -657,6 +836,11 @@ private:
|
||||
// Non-temporal streaming
|
||||
ACCESS_STREAM,
|
||||
};
|
||||
OrderedNode *LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0);
|
||||
OrderedNode *LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, OrderedNode *const Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, OrderedNode *const Src);
|
||||
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
+122
-29
@@ -35,19 +35,16 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W5 = _VExtractToGPR(16, 4, Src, 2);
|
||||
OrderedNode *NewVec{};
|
||||
NewVec = _VInsElement(16, 4, 3, 1, Dest, Dest);
|
||||
NewVec = _VInsElement(16, 4, 2, 0, NewVec, Dest);
|
||||
NewVec = _VInsElement(16, 4, 1, 3, NewVec, Src);
|
||||
NewVec = _VInsElement(16, 4, 0, 2, NewVec, Src);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, _Xor(W2, W0));
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, _Xor(W3, W1));
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, _Xor(W4, W2));
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, _Xor(W5, W3));
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
OrderedNode *Result = _VXor(16, 1, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -221,7 +218,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *XMM0 = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
@@ -263,47 +261,135 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Res0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode* OpDispatchBuilder::AESIMCImpl(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Res = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
return _VAESImc(Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode *Result = AESIMCImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESIMCOp(OpcodeArgs) {
|
||||
OrderedNode *Mixed = AESIMCImpl(Op);
|
||||
OrderedNode *Result = _VMov(16, Mixed);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Res = _VAESEnc(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
OrderedNode *Result = _VAESEnc(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Result = _VAESEnc(State, Key);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = _VMov(16, Result);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Res = _VAESEncLast(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
OrderedNode *Result = _VAESEncLast(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Result = _VAESEncLast(State, Key);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = _VMov(16, Result);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Res = _VAESDec(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
OrderedNode *Result = _VAESDec(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Result = _VAESDec(State, Key);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = _VMov(16, Result);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Res = _VAESDecLast(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
OrderedNode *Result = _VAESDecLast(Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Result = _VAESDecLast(State, Key);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = _VMov(16, Result);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
return _VAESKeyGenAssist(Src, RCON);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
OrderedNode *Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
auto Res = _VAESKeyGenAssist(Src, RCON);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
void OpDispatchBuilder::VAESKeyGenAssistOp(OpcodeArgs) {
|
||||
OrderedNode *Assist = AESKeyGenAssistImpl(Op);
|
||||
OrderedNode *Result = _VMov(16, Assist);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -320,11 +406,18 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(Src1, Src2, Selector);
|
||||
OrderedNode *Res = _PCLMUL(Src1, Src2, Selector);
|
||||
if (Is128Bit) {
|
||||
Res = _VMov(16, Res);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
|
||||
@@ -769,7 +769,11 @@ void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, Orde
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
|
||||
auto OpSize = SrcSize * 8;
|
||||
if (OpSize < Shift) {
|
||||
Shift &= (OpSize - 1);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, OpSize - Shift, Src1));
|
||||
}
|
||||
|
||||
// PF
|
||||
@@ -934,6 +938,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
|
||||
// If shift == 0, don't update flags
|
||||
@@ -963,7 +968,9 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode
|
||||
// OF
|
||||
{
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF);
|
||||
@@ -977,8 +984,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
|
||||
auto NewCF = _Bfe(1, OpSize - 1, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -989,8 +995,10 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Res), NewCF));
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1000,17 +1008,22 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, Ord
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, 0, Res);
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
}
|
||||
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is the top two MSBs XOR'd together
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1)));
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1686
-496
File diff suppressed because it is too large.
Load diff
@@ -782,6 +782,12 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -809,7 +815,8 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -854,6 +861,9 @@ void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
|
||||
auto sin = _F80SIN(a);
|
||||
auto cos = _F80COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -900,6 +910,9 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
OrderedNode *data = _VCastFromGPR(16, 8, low);
|
||||
data = _VInsGPR(16, 8, 1, data, high);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -1185,7 +1198,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
|
||||
// reset to default
|
||||
|
||||
@@ -778,6 +778,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -804,7 +810,8 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -831,6 +838,9 @@ void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
|
||||
auto sin = _F64SIN(a);
|
||||
auto cos = _F64COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -871,6 +881,9 @@ void OpDispatchBuilder::X87TANF64(OpcodeArgs) {
|
||||
|
||||
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(one, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -996,7 +1009,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
|
||||
// reset to default
|
||||
|
||||
@@ -266,10 +266,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x17, 1, X86InstInfo{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1E, 1, X86InstInfo{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1F, 1, X86InstInfo{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
{0x48, 8, X86InstInfo{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
@@ -283,8 +283,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
|
||||
@@ -24,8 +24,8 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
@@ -67,7 +67,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -76,7 +76,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -85,7 +85,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -94,7 +94,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -295,7 +295,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
|
||||
{0x20, 4, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x24, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SS", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2B, 1, X86InstInfo{"MOVNTSS", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x2C, 1, X86InstInfo{"CVTTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{0x2D, 1, X86InstInfo{"CVTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
@@ -305,18 +305,18 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 1, X86InstInfo{"RSQRTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x53, 1, X86InstInfo{"RCPSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 1, X86InstInfo{"RSQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x53, 1, X86InstInfo{"RCPSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x54, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSS2SD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSS2SD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5B, 1, X86InstInfo{"CVTTPS2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x60, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x68, 7, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -353,7 +353,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xD8, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xE0, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xE6, 1, X86InstInfo{"CVTDQ2PD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0xE6, 1, X86InstInfo{"CVTDQ2PD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0xE7, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xE8, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -375,7 +375,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x24, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SD", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2B, 1, X86InstInfo{"MOVNTSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x2C, 1, X86InstInfo{"CVTTSD2SI", TYPE_INST, GenFlagsSrcSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{0x2C, 1, X86InstInfo{"CVTTSD2SI", TYPE_INST, GenFlagsSrcSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{0x2D, 1, X86InstInfo{"CVTSD2SI", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{0x2E, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -383,16 +383,16 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSD2SS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5B, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x60, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
|
||||
+241
-242
@@ -17,71 +17,71 @@ void InitializeVEXTables() {
|
||||
static constexpr U16U8InfoStruct VEXTable[] = {
|
||||
// Map 0 (Reserved)
|
||||
// VEX Map 1
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x10), 1, X86InstInfo{"VMODUPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x10), 1, X86InstInfo{"VMOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x10), 1, X86InstInfo{"VMOVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x10), 1, X86InstInfo{"VMOVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x11), 1, X86InstInfo{"VMOVUPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x11), 1, X86InstInfo{"VMODUPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x11), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x11), 1, X86InstInfo{"VMOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x11), 1, X86InstInfo{"VMOVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x11), 1, X86InstInfo{"VMOVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x13), 1, X86InstInfo{"VMOVLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x13), 1, X86InstInfo{"VMOVLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x13), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x13), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x14), 1, X86InstInfo{"VUNPCKLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x14), 1, X86InstInfo{"VUNPCKLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x14), 1, X86InstInfo{"VUNPCKLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x14), 1, X86InstInfo{"VUNPCKLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x15), 1, X86InstInfo{"VUNPCKHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x15), 1, X86InstInfo{"VUNPCKHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x15), 1, X86InstInfo{"VUNPCKHPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x15), 1, X86InstInfo{"VUNPCKHPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x16), 1, X86InstInfo{"VMOVHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x16), 1, X86InstInfo{"VMOVHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x16), 1, X86InstInfo{"VMOVSHDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x16), 1, X86InstInfo{"VMOVHPS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x16), 1, X86InstInfo{"VMOVHPD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x16), 1, X86InstInfo{"VMOVSHDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x17), 1, X86InstInfo{"VMOVHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x17), 1, X86InstInfo{"VMOVHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x17), 1, X86InstInfo{"VMOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x17), 1, X86InstInfo{"VMOVHPD", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x50), 1, X86InstInfo{"VMOVMSKPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x50), 1, X86InstInfo{"VMOVMSKPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x50), 1, X86InstInfo{"VMOVMSKPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x50), 1, X86InstInfo{"VMOVMSKPD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x51), 1, X86InstInfo{"VSQRTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x51), 1, X86InstInfo{"VSQRTPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x51), 1, X86InstInfo{"VSQRTSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x51), 1, X86InstInfo{"VSQRTSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x51), 1, X86InstInfo{"VSQRTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x51), 1, X86InstInfo{"VSQRTPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x51), 1, X86InstInfo{"VSQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x51), 1, X86InstInfo{"VSQRTSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x52), 1, X86InstInfo{"VRSQRTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x52), 1, X86InstInfo{"VRSQRTSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x52), 1, X86InstInfo{"VRSQRTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x52), 1, X86InstInfo{"VRSQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x53), 1, X86InstInfo{"VRCPPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x53), 1, X86InstInfo{"VRCPSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x53), 1, X86InstInfo{"VRCPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x53), 1, X86InstInfo{"VRCPSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x54), 1, X86InstInfo{"VANDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x54), 1, X86InstInfo{"VANDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x54), 1, X86InstInfo{"VANDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x54), 1, X86InstInfo{"VANDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x55), 1, X86InstInfo{"VANDNPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x55), 1, X86InstInfo{"VANDNPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x55), 1, X86InstInfo{"VANDNPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x55), 1, X86InstInfo{"VANDNPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x56), 1, X86InstInfo{"VORPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x56), 1, X86InstInfo{"VORPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x56), 1, X86InstInfo{"VORPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x56), 1, X86InstInfo{"VORPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x57), 1, X86InstInfo{"VXORPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x57), 1, X86InstInfo{"VDORPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x57), 1, X86InstInfo{"VXORPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x57), 1, X86InstInfo{"VXORPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x60), 1, X86InstInfo{"VPUNPCKLBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x61), 1, X86InstInfo{"VPUNPCKLWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x62), 1, X86InstInfo{"VPUNPCKLDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x63), 1, X86InstInfo{"VPACKSSWB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x64), 1, X86InstInfo{"VPCMPGTB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x65), 1, X86InstInfo{"VPVMPGTW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x66), 1, X86InstInfo{"VPVMPGTD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x67), 1, X86InstInfo{"VPACKUSWB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x60), 1, X86InstInfo{"VPUNPCKLBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x61), 1, X86InstInfo{"VPUNPCKLWD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x62), 1, X86InstInfo{"VPUNPCKLDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x63), 1, X86InstInfo{"VPACKSSWB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x64), 1, X86InstInfo{"VPCMPGTB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x65), 1, X86InstInfo{"VPCMPGTW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x66), 1, X86InstInfo{"VPCMPGTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x67), 1, X86InstInfo{"VPACKUSWB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x70), 1, X86InstInfo{"VPSHUFD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x70), 1, X86InstInfo{"VPSHUFHW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -91,19 +91,19 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x72), 1, X86InstInfo{"", TYPE_VEX_GROUP_13, FLAGS_NONE, 0, nullptr}}, // VEX Group 13
|
||||
{OPD(1, 0b01, 0x73), 1, X86InstInfo{"", TYPE_VEX_GROUP_14, FLAGS_NONE, 0, nullptr}}, // VEX Group 14
|
||||
|
||||
{OPD(1, 0b01, 0x74), 1, X86InstInfo{"VPCMPEQB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x75), 1, X86InstInfo{"VPCMPEQW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x76), 1, X86InstInfo{"VPCMPEQD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x74), 1, X86InstInfo{"VPCMPEQB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x75), 1, X86InstInfo{"VPCMPEQW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x76), 1, X86InstInfo{"VPCMPEQD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x77), 1, X86InstInfo{"VZERO*", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x77), 1, X86InstInfo{"VZERO*", TYPE_INST, GenFlagsDstSize(SIZE_128BIT), 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xC2), 1, X86InstInfo{"VCMPccPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xC2), 1, X86InstInfo{"VCMPccPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0xC2), 1, X86InstInfo{"VCMPccSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xC2), 1, X86InstInfo{"VCMPccSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0xC2), 1, X86InstInfo{"VCMPccPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, 0b01, 0xC2), 1, X86InstInfo{"VCMPccPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, 0b10, 0xC2), 1, X86InstInfo{"VCMPccSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, 0b11, 0xC2), 1, X86InstInfo{"VCMPccSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xC4), 1, X86InstInfo{"VPINSRW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xC5), 1, X86InstInfo{"VPEXTRW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xC5), 1, X86InstInfo{"VPEXTRW", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xC6), 1, X86InstInfo{"VSHUFPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xC6), 1, X86InstInfo{"VSHUFPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -112,167 +112,166 @@ void InitializeVEXTables() {
|
||||
// This table doesn't state which VEX.pp is for which instruction
|
||||
// XXX: Confirm all the above encoding opcodes
|
||||
|
||||
{OPD(1, 0b00, 0x28), 1, X86InstInfo{"VMOVAPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x28), 1, X86InstInfo{"VMOVAPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x28), 1, X86InstInfo{"VMOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x28), 1, X86InstInfo{"VMOVAPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x29), 1, X86InstInfo{"VMOVAPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x29), 1, X86InstInfo{"VMOVAPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x29), 1, X86InstInfo{"VMOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x29), 1, X86InstInfo{"VMOVAPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, X86InstInfo{"VCVTSI2SS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2A), 1, X86InstInfo{"VCVTSI2SD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, X86InstInfo{"VMOVNTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2B), 1, X86InstInfo{"VMOVNTPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x2B), 1, X86InstInfo{"VMOVNTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2B), 1, X86InstInfo{"VMOVNTPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2C), 1, X86InstInfo{"VCVTTSS2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2C), 1, X86InstInfo{"VCVTTSD2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x2C), 1, X86InstInfo{"VCVTTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2C), 1, X86InstInfo{"VCVTTSD2SI", TYPE_INST, GenFlagsSrcSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2D), 1, X86InstInfo{"VCVTSS2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2D), 1, X86InstInfo{"VCVTSD2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x2D), 1, X86InstInfo{"VCVTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2D), 1, X86InstInfo{"VCVTSD2SI", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x2E), 1, X86InstInfo{"VUCOMISS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2E), 1, X86InstInfo{"VUCOMISD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x2E), 1, X86InstInfo{"VUCOMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2E), 1, X86InstInfo{"VUCOMISD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x2F), 1, X86InstInfo{"VUCOMISS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2F), 1, X86InstInfo{"VUCOMISD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x2F), 1, X86InstInfo{"VCOMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2F), 1, X86InstInfo{"VCOMISD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, X86InstInfo{"VADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x58), 1, X86InstInfo{"VADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x58), 1, X86InstInfo{"VADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x58), 1, X86InstInfo{"VADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x58), 1, X86InstInfo{"VADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x58), 1, X86InstInfo{"VADDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x58), 1, X86InstInfo{"VADDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x58), 1, X86InstInfo{"VADDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x59), 1, X86InstInfo{"VMULPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x59), 1, X86InstInfo{"VMULPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x59), 1, X86InstInfo{"VMULSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x59), 1, X86InstInfo{"VMULSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x59), 1, X86InstInfo{"VMULPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x59), 1, X86InstInfo{"VMULPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x59), 1, X86InstInfo{"VMULSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x59), 1, X86InstInfo{"VMULSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, X86InstInfo{"VCVTDQ2PS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5B), 1, X86InstInfo{"VCVTPS2DQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5B), 1, X86InstInfo{"VCVTPS2DQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x5B), 1, X86InstInfo{"VCVTDQ2PS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5B), 1, X86InstInfo{"VCVTPS2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5B), 1, X86InstInfo{"VCVTTPS2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, X86InstInfo{"VSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5C), 1, X86InstInfo{"VSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5C), 1, X86InstInfo{"VSUBSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5C), 1, X86InstInfo{"VSUBSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x5C), 1, X86InstInfo{"VSUBPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5C), 1, X86InstInfo{"VSUBPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5C), 1, X86InstInfo{"VSUBSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5C), 1, X86InstInfo{"VSUBSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x5D), 1, X86InstInfo{"VMINPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5D), 1, X86InstInfo{"VMINPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5D), 1, X86InstInfo{"VMINSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5D), 1, X86InstInfo{"VMINSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x5D), 1, X86InstInfo{"VMINPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5D), 1, X86InstInfo{"VMINPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5D), 1, X86InstInfo{"VMINSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5D), 1, X86InstInfo{"VMINSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x5E), 1, X86InstInfo{"VDIVPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5E), 1, X86InstInfo{"VDIVPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5E), 1, X86InstInfo{"VDIVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5E), 1, X86InstInfo{"VDIVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x5E), 1, X86InstInfo{"VDIVPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5E), 1, X86InstInfo{"VDIVPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5E), 1, X86InstInfo{"VDIVSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5E), 1, X86InstInfo{"VDIVSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x5F), 1, X86InstInfo{"VMAXPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5F), 1, X86InstInfo{"VMAXPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5F), 1, X86InstInfo{"VMAXSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5F), 1, X86InstInfo{"VMAXSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x5F), 1, X86InstInfo{"VMAXPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x5F), 1, X86InstInfo{"VMAXPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x5F), 1, X86InstInfo{"VMAXSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x5F), 1, X86InstInfo{"VMAXSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
|
||||
{OPD(1, 0b01, 0x68), 1, X86InstInfo{"VPUNPCKHBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x69), 1, X86InstInfo{"VPUNPCKHWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6A), 1, X86InstInfo{"VPUNPCKHDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6B), 1, X86InstInfo{"VPACKSSDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6C), 1, X86InstInfo{"VPUNPCKLQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6D), 1, X86InstInfo{"VPUNPCKHQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x68), 1, X86InstInfo{"VPUNPCKHBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x69), 1, X86InstInfo{"VPUNPCKHWD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6A), 1, X86InstInfo{"VPUNPCKHDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6B), 1, X86InstInfo{"VPACKSSDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6C), 1, X86InstInfo{"VPUNPCKLQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6D), 1, X86InstInfo{"VPUNPCKHQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6E), 1, X86InstInfo{"VMOV*", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x6F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x6F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x6F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7C), 1, X86InstInfo{"VHADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7C), 1, X86InstInfo{"VHADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7C), 1, X86InstInfo{"VHADDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7C), 1, X86InstInfo{"VHADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7D), 1, X86InstInfo{"VHSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7D), 1, X86InstInfo{"VHSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7E), 1, X86InstInfo{"VMOV*", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7E), 1, X86InstInfo{"VMOVQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7E), 1, X86InstInfo{"VMOV*", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7E), 1, X86InstInfo{"VMOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
{OPD(1, 0b01, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
{OPD(1, 0b10, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
{OPD(1, 0b11, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
|
||||
{OPD(1, 0b01, 0xD0), 1, X86InstInfo{"VADDSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xD0), 1, X86InstInfo{"VADDSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD0), 1, X86InstInfo{"VADDSUBPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xD0), 1, X86InstInfo{"VADDSUBPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xD1), 1, X86InstInfo{"VPSRLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD2), 1, X86InstInfo{"VPSRLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD3), 1, X86InstInfo{"VPSRLQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD4), 1, X86InstInfo{"VPADDQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD5), 1, X86InstInfo{"VPMULLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD6), 1, X86InstInfo{"VMOVQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD7), 1, X86InstInfo{"VPMOVMSKB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD1), 1, X86InstInfo{"VPSRLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD2), 1, X86InstInfo{"VPSRLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD3), 1, X86InstInfo{"VPSRLQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD4), 1, X86InstInfo{"VPADDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD5), 1, X86InstInfo{"VPMULLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD6), 1, X86InstInfo{"VMOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD7), 1, X86InstInfo{"VPMOVMSKB", TYPE_UNDEC, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xD8), 1, X86InstInfo{"VPSUBUSB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD9), 1, X86InstInfo{"VPSUBUSW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDA), 1, X86InstInfo{"VPMINUB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDB), 1, X86InstInfo{"VPAND", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDC), 1, X86InstInfo{"VPADDUSB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDD), 1, X86InstInfo{"VPADDUSW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDE), 1, X86InstInfo{"VPMAXUB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDF), 1, X86InstInfo{"VPANDN", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD8), 1, X86InstInfo{"VPSUBUSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD9), 1, X86InstInfo{"VPSUBUSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDA), 1, X86InstInfo{"VPMINUB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDB), 1, X86InstInfo{"VPAND", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDC), 1, X86InstInfo{"VPADDUSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDD), 1, X86InstInfo{"VPADDUSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDE), 1, X86InstInfo{"VPMAXUB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDF), 1, X86InstInfo{"VPANDN", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE0), 1, X86InstInfo{"VPAVGB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE1), 1, X86InstInfo{"VPSRAW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE2), 1, X86InstInfo{"VPSRAD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE3), 1, X86InstInfo{"VPAVGW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE4), 1, X86InstInfo{"VPMULHUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE5), 1, X86InstInfo{"VPMULHW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE0), 1, X86InstInfo{"VPAVGB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE1), 1, X86InstInfo{"VPSRAW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE2), 1, X86InstInfo{"VPSRAD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE3), 1, X86InstInfo{"VPAVGW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE4), 1, X86InstInfo{"VPMULHUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE5), 1, X86InstInfo{"VPMULHW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, X86InstInfo{"VCVTTPD2DQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0xE6), 1, X86InstInfo{"VCVTDQ2PD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xE6), 1, X86InstInfo{"VCVTPD2DQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE6), 1, X86InstInfo{"VCVTTPD2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0xE6), 1, X86InstInfo{"VCVTDQ2PD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xE6), 1, X86InstInfo{"VCVTPD2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE7), 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE8), 1, X86InstInfo{"VPSUBSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE9), 1, X86InstInfo{"VPSUBSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEA), 1, X86InstInfo{"VPMINSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEB), 1, X86InstInfo{"VPOR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEC), 1, X86InstInfo{"VPADDSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xED), 1, X86InstInfo{"VPADDSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEE), 1, X86InstInfo{"VPMAXSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEF), 1, X86InstInfo{"VPXOR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE8), 1, X86InstInfo{"VPSUBSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE9), 1, X86InstInfo{"VPSUBSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEA), 1, X86InstInfo{"VPMINSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEB), 1, X86InstInfo{"VPOR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEC), 1, X86InstInfo{"VPADDSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xED), 1, X86InstInfo{"VPADDSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEE), 1, X86InstInfo{"VPMAXSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEF), 1, X86InstInfo{"VPXOR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b11, 0xF0), 1, X86InstInfo{"VLDDQU", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xF0), 1, X86InstInfo{"VLDDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xF1), 1, X86InstInfo{"VPSLLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF2), 1, X86InstInfo{"VPSLLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF3), 1, X86InstInfo{"VPSLLQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF4), 1, X86InstInfo{"VPMULUDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF1), 1, X86InstInfo{"VPSLLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF2), 1, X86InstInfo{"VPSLLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF3), 1, X86InstInfo{"VPSLLQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF4), 1, X86InstInfo{"VPMULUDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF5), 1, X86InstInfo{"VPMADDWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF6), 1, X86InstInfo{"VPSADBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF7), 1, X86InstInfo{"VMASKMOVDQU", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF7), 1, X86InstInfo{"VMASKMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xF8), 1, X86InstInfo{"VPSUBB", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF9), 1, X86InstInfo{"VPSUBW", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFA), 1, X86InstInfo{"VPSUBD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFB), 1, X86InstInfo{"VPSUBQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFC), 1, X86InstInfo{"VPADDB", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFD), 1, X86InstInfo{"VPADDW", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFE), 1, X86InstInfo{"VPADDD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF8), 1, X86InstInfo{"VPSUBB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xF9), 1, X86InstInfo{"VPSUBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFA), 1, X86InstInfo{"VPSUBD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFB), 1, X86InstInfo{"VPSUBQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFC), 1, X86InstInfo{"VPADDB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFD), 1, X86InstInfo{"VPADDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFE), 1, X86InstInfo{"VPADDD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
// VEX Map 2
|
||||
{OPD(2, 0b01, 0x00), 1, X86InstInfo{"VPSHUFB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x01), 1, X86InstInfo{"VPADDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x02), 1, X86InstInfo{"VPHADDD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x00), 1, X86InstInfo{"VPSHUFB", TYPE_UNDEC, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x01), 1, X86InstInfo{"VPHADDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x02), 1, X86InstInfo{"VPHADDD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x03), 1, X86InstInfo{"VPHADDSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x04), 1, X86InstInfo{"VPMADDUBSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x05), 1, X86InstInfo{"VPHSUBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x06), 1, X86InstInfo{"VPHSUBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x05), 1, X86InstInfo{"VPHSUBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x06), 1, X86InstInfo{"VPHSUBD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x07), 1, X86InstInfo{"VPHSUBSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x08), 1, X86InstInfo{"VPSIGNB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x09), 1, X86InstInfo{"VPSIGNW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0A), 1, X86InstInfo{"VPSIGND", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0B), 1, X86InstInfo{"VPMULHRSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x08), 1, X86InstInfo{"VPSIGNB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x09), 1, X86InstInfo{"VPSIGNW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0A), 1, X86InstInfo{"VPSIGND", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0B), 1, X86InstInfo{"VPMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0C), 1, X86InstInfo{"VPERMILPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0D), 1, X86InstInfo{"VPERMILPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0E), 1, X86InstInfo{"VTESTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -282,59 +281,59 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b01, 0x16), 1, X86InstInfo{"VPERMPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x17), 1, X86InstInfo{"VPTEST", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x18), 1, X86InstInfo{"VBROADCASTSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x19), 1, X86InstInfo{"VBROADCASTSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1A), 1, X86InstInfo{"VBROADCASTF128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1C), 1, X86InstInfo{"VPABSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1D), 1, X86InstInfo{"VPABSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1E), 1, X86InstInfo{"VPABSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x18), 1, X86InstInfo{"VBROADCASTSS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x19), 1, X86InstInfo{"VBROADCASTSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1A), 1, X86InstInfo{"VBROADCASTF128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1C), 1, X86InstInfo{"VPABSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1D), 1, X86InstInfo{"VPABSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x1E), 1, X86InstInfo{"VPABSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x20), 1, X86InstInfo{"VPMOVSXBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x21), 1, X86InstInfo{"VPMOVSXBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x22), 1, X86InstInfo{"VPMOVSXBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x23), 1, X86InstInfo{"VPMOVSXWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x24), 1, X86InstInfo{"VPMOVSXWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x25), 1, X86InstInfo{"VPMOVSXDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x20), 1, X86InstInfo{"VPMOVSXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x21), 1, X86InstInfo{"VPMOVSXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x22), 1, X86InstInfo{"VPMOVSXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x23), 1, X86InstInfo{"VPMOVSXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x24), 1, X86InstInfo{"VPMOVSXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x25), 1, X86InstInfo{"VPMOVSXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x28), 1, X86InstInfo{"VPMULDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x29), 1, X86InstInfo{"VPCMPEQQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2A), 1, X86InstInfo{"VMOVNTDQA", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2B), 1, X86InstInfo{"VPACKUSDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x28), 1, X86InstInfo{"VPMULDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x29), 1, X86InstInfo{"VPCMPEQQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2A), 1, X86InstInfo{"VMOVNTDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2B), 1, X86InstInfo{"VPACKUSDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2C), 1, X86InstInfo{"VMASKMOVPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2D), 1, X86InstInfo{"VMASKMOVPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2E), 1, X86InstInfo{"VMASKMOVPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2F), 1, X86InstInfo{"VMASKMOVPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x30), 1, X86InstInfo{"VPMOVZXBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x31), 1, X86InstInfo{"VPMOVZXBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x32), 1, X86InstInfo{"VPMOVZXBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x33), 1, X86InstInfo{"VPMOVZXWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x34), 1, X86InstInfo{"VPMOVZXWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x35), 1, X86InstInfo{"VPMOVZXDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x30), 1, X86InstInfo{"VPMOVZXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x31), 1, X86InstInfo{"VPMOVZXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x32), 1, X86InstInfo{"VPMOVZXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x33), 1, X86InstInfo{"VPMOVZXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x34), 1, X86InstInfo{"VPMOVZXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x35), 1, X86InstInfo{"VPMOVZXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x36), 1, X86InstInfo{"VPERMD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x37), 1, X86InstInfo{"VPVMPGTQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x37), 1, X86InstInfo{"VPCMPGTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x38), 1, X86InstInfo{"VPMINSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x39), 1, X86InstInfo{"VPMINSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3A), 1, X86InstInfo{"VPMINUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3B), 1, X86InstInfo{"VPMINUD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3C), 1, X86InstInfo{"VPMAXSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3D), 1, X86InstInfo{"VPMAXSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3E), 1, X86InstInfo{"VPMAXUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3F), 1, X86InstInfo{"VPMAXUD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x38), 1, X86InstInfo{"VPMINSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x39), 1, X86InstInfo{"VPMINSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3A), 1, X86InstInfo{"VPMINUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3B), 1, X86InstInfo{"VPMINUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3C), 1, X86InstInfo{"VPMAXSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3D), 1, X86InstInfo{"VPMAXSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3E), 1, X86InstInfo{"VPMAXUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x3F), 1, X86InstInfo{"VPMAXUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x40), 1, X86InstInfo{"VPMULLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x41), 1, X86InstInfo{"VPHMINPOSUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x40), 1, X86InstInfo{"VPMULLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x41), 1, X86InstInfo{"VPHMINPOSUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x45), 1, X86InstInfo{"VPSRLV", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBBROADCASTI128", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x78), 1, X86InstInfo{"VPBROADCASTB", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x79), 1, X86InstInfo{"VPBROADCASTW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x78), 1, X86InstInfo{"VPBROADCASTB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x79), 1, X86InstInfo{"VPBROADCASTW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x8C), 1, X86InstInfo{"VPMASKMOV", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x8E), 1, X86InstInfo{"VPMASKMOV", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -380,11 +379,11 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b01, 0xB6), 1, X86InstInfo{"VFMADDSUB231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB7), 1, X86InstInfo{"VFMSUBADD231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0xDB), 1, X86InstInfo{"VAESIMC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDC), 1, X86InstInfo{"VAESENC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDD), 1, X86InstInfo{"VAESENCLAST", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDE), 1, X86InstInfo{"VAESDEC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDF), 1, X86InstInfo{"VAESDECLAST", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDB), 1, X86InstInfo{"VAESIMC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDC), 1, X86InstInfo{"VAESENC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDD), 1, X86InstInfo{"VAESENCLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDE), 1, X86InstInfo{"VAESDEC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDF), 1, X86InstInfo{"VAESDECLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b00, 0xF2), 1, X86InstInfo{"ANDN", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
|
||||
@@ -406,43 +405,43 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b11, 0xF7), 1, X86InstInfo{"SHRX", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_2ND_SRC, 0, nullptr}},
|
||||
|
||||
// VEX Map 3
|
||||
{OPD(3, 0b01, 0x00), 1, X86InstInfo{"VPERMQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x01), 1, X86InstInfo{"VPERMPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x00), 1, X86InstInfo{"VPERMQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x01), 1, X86InstInfo{"VPERMPD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x02), 1, X86InstInfo{"VPBLENDD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x04), 1, X86InstInfo{"VPERMILPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x05), 1, X86InstInfo{"VPERMILPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x06), 1, X86InstInfo{"VPERM2F128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x04), 1, X86InstInfo{"VPERMILPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x05), 1, X86InstInfo{"VPERMILPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x06), 1, X86InstInfo{"VPERM2F128", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x08), 1, X86InstInfo{"VROUNDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x09), 1, X86InstInfo{"VROUNDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x0A), 1, X86InstInfo{"VROUNDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x0B), 1, X86InstInfo{"VROUNDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x08), 1, X86InstInfo{"VROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x09), 1, X86InstInfo{"VROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x0A), 1, X86InstInfo{"VROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x0B), 1, X86InstInfo{"VROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x0C), 1, X86InstInfo{"VBLENDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x0D), 1, X86InstInfo{"VBLENDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x0E), 1, X86InstInfo{"VBLENDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x0F), 1, X86InstInfo{"VPALIGNR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x14), 1, X86InstInfo{"VPEXTRB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x15), 1, X86InstInfo{"VPEXTRW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x16), 1, X86InstInfo{"VPEXTRD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x17), 1, X86InstInfo{"VEXTRACTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x14), 1, X86InstInfo{"VPEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x15), 1, X86InstInfo{"VPEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x16), 1, X86InstInfo{"VPEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x17), 1, X86InstInfo{"VEXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x18), 1, X86InstInfo{"VINSERTF128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x18), 1, X86InstInfo{"VINSERTF128", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x19), 1, X86InstInfo{"VEXTRACTF128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x1D), 1, X86InstInfo{"VCVTPS2PH", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x20), 1, X86InstInfo{"VPINSRB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x21), 1, X86InstInfo{"VINSERTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x21), 1, X86InstInfo{"VINSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x22), 1, X86InstInfo{"VPINSRD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x38), 1, X86InstInfo{"VINSERTI128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x38), 1, X86InstInfo{"VINSERTI128", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x39), 1, X86InstInfo{"VEXTRACTI128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x40), 1, X86InstInfo{"VDPPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x41), 1, X86InstInfo{"VDPPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x40), 1, X86InstInfo{"VDPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x41), 1, X86InstInfo{"VDPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x42), 1, X86InstInfo{"VMPSADBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x44), 1, X86InstInfo{"VPCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x46), 1, X86InstInfo{"VPERM2I128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x46), 1, X86InstInfo{"VPERM2I128", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x48), 1, X86InstInfo{"VPERMILzz2PS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x49), 1, X86InstInfo{"VPERMILzz2PD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -478,7 +477,7 @@ void InitializeVEXTables() {
|
||||
{OPD(3, 0b01, 0x7E), 1, X86InstInfo{"VFNMSUBSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7F), 1, X86InstInfo{"VFNMSUBSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, X86InstInfo{"VAESKEYGENASSIST", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0xDF), 1, X86InstInfo{"VAESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b11, 0xF0), 1, X86InstInfo{"RORX", TYPE_INST, FLAGS_MODRM, 1, nullptr}},
|
||||
|
||||
@@ -488,21 +487,21 @@ void InitializeVEXTables() {
|
||||
|
||||
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
static constexpr U8U8InfoStruct VEXGroupTable[] = {
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b010), 1, X86InstInfo{"VPSRLW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b100), 1, X86InstInfo{"VPSRAW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b110), 1, X86InstInfo{"VPSLLW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b010), 1, X86InstInfo{"VPSRLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b100), 1, X86InstInfo{"VPSRAW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b110), 1, X86InstInfo{"VPSLLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(TYPE_VEX_GROUP_13, 1, 0b010), 1, X86InstInfo{"VPSRLD", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_13, 1, 0b100), 1, X86InstInfo{"VPSRAD", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_13, 1, 0b110), 1, X86InstInfo{"VPSLLD", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_13, 1, 0b010), 1, X86InstInfo{"VPSRLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_13, 1, 0b100), 1, X86InstInfo{"VPSRAD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_13, 1, 0b110), 1, X86InstInfo{"VPSLLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b010), 1, X86InstInfo{"VPSRLQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b011), 1, X86InstInfo{"VPSRLDQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b110), 1, X86InstInfo{"VPSLLQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b111), 1, X86InstInfo{"VPSLLDQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b010), 1, X86InstInfo{"VPSRLQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b011), 1, X86InstInfo{"VPSRLDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b110), 1, X86InstInfo{"VPSLLQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_14, 1, 0b111), 1, X86InstInfo{"VPSLLDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(TYPE_VEX_GROUP_15, 1, 0b010), 1, X86InstInfo{"VLDMXCSR", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_15, 1, 0b011), 1, X86InstInfo{"VSTMXCSR", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_15, 0, 0b010), 1, X86InstInfo{"VLDMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_15, 0, 0b011), 1, X86InstInfo{"VSTMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_VEX_GROUP_17, 0, 0b001), 1, X86InstInfo{"BLSR", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_DST, 0, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_17, 0, 0b010), 1, X86InstInfo{"BLSMSK", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_DST, 0, nullptr}},
|
||||
|
||||
+15
-2
@@ -167,6 +167,8 @@ namespace FEXCore {
|
||||
* address to another function. The original callee address is passed
|
||||
* to the target function through an implicit argument stored in r11.
|
||||
*
|
||||
* For 32-bit the implicit argument is stored in the lower 32-bits of mm0.
|
||||
*
|
||||
* The primary use case of this is ensuring that host function pointers
|
||||
* returned from thunked APIs can safely be called by the guest.
|
||||
*/
|
||||
@@ -199,7 +201,12 @@ namespace FEXCore {
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
emit->_StoreContext(GPRSize, IR::GPRClass, emit->_Constant(Entrypoint), offsetof(Core::CPUState, gregs[X86State::REG_R11]));
|
||||
if (GPRSize == 8) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
}
|
||||
else {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, mm[0][0]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
@@ -263,7 +270,10 @@ namespace FEXCore {
|
||||
|
||||
auto Name = Args->Name;
|
||||
|
||||
auto SOName = CTX->Config.ThunkHostLibsPath() + "/" + (const char*)Name + "-host.so";
|
||||
auto SOName = (CTX->Config.Is64BitMode() ?
|
||||
CTX->Config.ThunkHostLibsPath() :
|
||||
CTX->Config.ThunkHostLibsPath32())
|
||||
+ "/" + (const char*)Name + "-host.so";
|
||||
|
||||
LogMan::Msg::DFmt("LoadLib: {} -> {}", Name, SOName);
|
||||
|
||||
@@ -431,6 +441,9 @@ namespace FEXCore {
|
||||
|
||||
FEX_DEFAULT_VISIBILITY
|
||||
void FinalizeHostTrampolineForGuestFunction(HostToGuestTrampolinePtr* TrampolineAddress, void* HostPacker) {
|
||||
|
||||
if (TrampolineAddress == nullptr) return;
|
||||
|
||||
auto& Trampoline = GetInstanceInfo(TrampolineAddress);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Trampoline.CallCallback == (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
|
||||
+39
-85
@@ -302,10 +302,6 @@
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
"GPR = Mov GPR:$Value": {
|
||||
"DestSize": "GetOpSize(_Value)"
|
||||
},
|
||||
|
||||
"GPR = ExtractElementPair GPRPair:$Pair, u8:$Element": {
|
||||
"Desc": ["Extracts a register for the register pair"],
|
||||
"DestSize": "GetOpSize(_Pair) >> 1"
|
||||
@@ -359,22 +355,26 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContext u8:#ByteSize, RegisterClass:$Class, SSA:$Value, u32:$Offset": {
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -385,7 +385,9 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
"StoreContextIndexed SSA:$Value, GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
|
||||
@@ -397,7 +399,9 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -475,22 +479,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VLoadMemElement u8:#RegisterSize, u8:#ElementSize, FPR:$Value, GPR:$Addr, u8:$Index, u8:$Align{1}": {
|
||||
"Desc": ["Loads an element of size #ElementSize in to $Value from $Addr at $Index"
|
||||
],
|
||||
"OpClass": "Memory",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"VStoreMemElement u8:#RegisterSize, u8:#ElementSize, FPR:$Value, GPR:$Addr, u8:$Index, u8:$Align": {
|
||||
"Desc": ["Stores an element of size #ElementSize from $Value[$Index] to $Addr"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ElementSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"CacheLineClear GPR:$Addr": {
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified",
|
||||
"Only clears the data cachelines. Doesn't do any zeroing"
|
||||
@@ -845,27 +833,25 @@
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))"
|
||||
},
|
||||
"GPR = Extr GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
"Valid LSB range is 0-31 for 32bit and 0-63 for 64bit",
|
||||
"<Size * 2> ConcatValue = $Upper:$Lower",
|
||||
"Result = ConcatValue<LSB+Size - 1: LSB>"
|
||||
]
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
"Valid LSB range is 0-31 for 32bit and 0-63 for 64bit",
|
||||
"<Size * 2> ConcatValue = $Upper:$Lower",
|
||||
"Result = ConcatValue<LSB+Size - 1: LSB>"
|
||||
]
|
||||
},
|
||||
"GPR = PDep GPR:$Input, GPR:$Mask": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
]
|
||||
"Desc": ["Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = PExt GPR:$Input, GPR:$Mask": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
]
|
||||
"Desc": ["Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = LDiv GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
@@ -924,15 +910,6 @@
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
"FPR = SplatVector2 FPR:$Scalar": {
|
||||
"NumElements": "2",
|
||||
"DestSize": "GetOpSize(_Scalar) * 2"
|
||||
},
|
||||
"FPR = SplatVector4 FPR:$Scalar": {
|
||||
"NumElements": "4",
|
||||
"DestSize": "GetOpSize(_Scalar) * 4"
|
||||
},
|
||||
|
||||
"FPR = VMov u8:#RegisterSize, FPR:$Source": {
|
||||
"Desc" : ["Copy vector register",
|
||||
"When Register size is smaller than Source register size,",
|
||||
@@ -941,12 +918,6 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VBitcast u8:#RegisterSize, u8:#ElementSize, FPR:$Source": {
|
||||
"Desc": ["Workaround for issue with LLVM breaking when loading scalar elements to vectors"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VectorZero u8:#RegisterSize": {
|
||||
"Desc": ["Generates a vector zero",
|
||||
"Useful to generate a zero vector without any previous dependencies"
|
||||
@@ -1038,24 +1009,11 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VExtractElement u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"DestSize": "ElementSize"
|
||||
},
|
||||
"FPR = VDupElement u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"Desc": ["Duplicates one element from the source register across the whole register"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSLI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$ByteShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VSRI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$ByteShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VShlI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1089,7 +1047,7 @@
|
||||
},
|
||||
"FPR = VSXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1101,7 +1059,7 @@
|
||||
},
|
||||
"FPR = VUXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Zero extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1124,9 +1082,9 @@
|
||||
},
|
||||
|
||||
"FPR = VRev64 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1320,10 +1278,6 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsScalarElement u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, FPR:$DestVector, FPR:$SrcScalar": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsGPR u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, FPR:$DestVector, GPR:$Src": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1587,9 +1541,9 @@
|
||||
"DestSize": "16"
|
||||
},
|
||||
"GPR = F80Cmp FPR:$X80Src1, FPR:$X80Src2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"FPR = F80BCDLoad FPR:$X80Src": {
|
||||
|
||||
@@ -162,6 +162,12 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
else if (Arg == "FPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRClass};
|
||||
}
|
||||
else if (Arg == "GPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRFixedClass};
|
||||
}
|
||||
else if (Arg == "FPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRFixedClass};
|
||||
}
|
||||
else if (Arg == "GPRPair") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRPairClass};
|
||||
}
|
||||
|
||||
+3
-9
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
@@ -36,15 +37,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineCo
|
||||
|
||||
InsertPass(CreateSyscallOptimization());
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
else {
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
@@ -66,6 +58,8 @@ void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAV
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
bool Changed = false;
|
||||
for (auto const &Pass : Passes) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
|
||||
@@ -21,7 +21,6 @@ std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusiveP
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
|
||||
bool OptimizeSRA,
|
||||
bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
namespace Validation {
|
||||
|
||||
+12
-9
@@ -6,7 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
|
||||
#if defined(_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
//aarch64 heuristics
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
@@ -20,6 +20,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
@@ -45,20 +46,20 @@ uint64_t getMask(IROp_Header* Op) {
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
// very lazy heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; }
|
||||
static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8;
|
||||
}
|
||||
#elif defined(_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
//aarch64 heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { if (width < 32) width = 32; return vixl::aarch64::Assembler::IsImmLogical(imm, width); }
|
||||
static bool IsImmAddSub(uint64_t imm) { return vixl::aarch64::Assembler::IsImmAddSub(imm); }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == AccessSize;
|
||||
}
|
||||
#elif JIT_X86_64
|
||||
// very lazy heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; }
|
||||
static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8;
|
||||
}
|
||||
#else
|
||||
#error No inline constant heuristics for this target
|
||||
#endif
|
||||
@@ -1028,6 +1029,8 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
|
||||
bool ConstProp::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
@@ -22,6 +23,7 @@ private:
|
||||
};
|
||||
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
|
||||
|
||||
+134
-66
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
@@ -76,24 +77,6 @@ namespace {
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // SSE padding in non-AVX case
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
@@ -102,7 +85,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
sizeof(FEXCore::Core::CPUState::rip),
|
||||
},
|
||||
DefaultAccess[0],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -112,62 +95,134 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
|
||||
FEXCore::Core::CPUState::GPR_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[1],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es),
|
||||
sizeof(FEXCore::Core::CPUState::es),
|
||||
offsetof(FEXCore::Core::CPUState, es_idx),
|
||||
sizeof(FEXCore::Core::CPUState::es_idx),
|
||||
},
|
||||
DefaultAccess[2],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs),
|
||||
sizeof(FEXCore::Core::CPUState::cs),
|
||||
offsetof(FEXCore::Core::CPUState, cs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::cs_idx),
|
||||
},
|
||||
DefaultAccess[3],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss),
|
||||
sizeof(FEXCore::Core::CPUState::ss),
|
||||
offsetof(FEXCore::Core::CPUState, ss_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ss_idx),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds),
|
||||
sizeof(FEXCore::Core::CPUState::ds),
|
||||
offsetof(FEXCore::Core::CPUState, ds_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ds_idx),
|
||||
},
|
||||
DefaultAccess[5],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::gs_idx),
|
||||
},
|
||||
DefaultAccess[6],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
sizeof(FEXCore::Core::CPUState::fs),
|
||||
offsetof(FEXCore::Core::CPUState, fs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::fs_idx),
|
||||
},
|
||||
DefaultAccess[7],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad),
|
||||
sizeof(FEXCore::Core::CPUState::_pad),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es_cached),
|
||||
sizeof(FEXCore::Core::CPUState::es_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::cs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ss_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ds_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::gs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::fs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -178,7 +233,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -189,7 +244,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -199,7 +254,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
DefaultAccess[9],
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -210,7 +265,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
|
||||
FEXCore::Core::CPUState::FLAG_SIZE,
|
||||
},
|
||||
DefaultAccess[10],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -221,7 +276,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
|
||||
FEXCore::Core::CPUState::MM_REG_SIZE
|
||||
},
|
||||
DefaultAccess[11],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -233,7 +288,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::gdt[0]),
|
||||
},
|
||||
DefaultAccess[12],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -244,7 +299,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FCW),
|
||||
sizeof(FEXCore::Core::CPUState::FCW),
|
||||
},
|
||||
DefaultAccess[13],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -254,7 +309,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FTW),
|
||||
sizeof(FEXCore::Core::CPUState::FTW),
|
||||
},
|
||||
DefaultAccess[14],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -288,40 +343,55 @@ namespace {
|
||||
ContextClassification->at(Offset).StoreNode = nullptr;
|
||||
};
|
||||
size_t Offset = 0;
|
||||
SetAccess(Offset++, DefaultAccess[0]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[1]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[2]);
|
||||
SetAccess(Offset++, DefaultAccess[3]);
|
||||
SetAccess(Offset++, DefaultAccess[4]);
|
||||
SetAccess(Offset++, DefaultAccess[5]);
|
||||
SetAccess(Offset++, DefaultAccess[6]);
|
||||
SetAccess(Offset++, DefaultAccess[7]);
|
||||
// Segment indexes
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
// Segments
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad2
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[8]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
if (!SupportsAVX) {
|
||||
SetAccess(Offset++, DefaultAccess[9]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[10]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[11]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[12]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
SetAccess(Offset++, DefaultAccess[14]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
@@ -449,7 +519,6 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa29 i128 = VBitcast %ssa26 i128
|
||||
*
|
||||
* eg.
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
@@ -462,13 +531,11 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* eg.
|
||||
* (%%ssa189) StoreContext %ssa188 i128, 0x10, 0xa0
|
||||
* %ssa190 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa191 i128 = VBitcast %ssa188 i128
|
||||
* %ssa192 i128 = VAdd %ssa191 i128, %ssa190 i128, 0x10, 0x4
|
||||
* %ssa192 i128 = VAdd %ssa188 i128, %ssa190 i128, 0x10, 0x4
|
||||
* (%%ssa193) StoreContext %ssa192 i128, 0x10, 0xa0
|
||||
* Converts to
|
||||
* %ssa173 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa174 i128 = VBitcast %ssa172 i128
|
||||
* %ssa175 i128 = VAdd %ssa174 i128, %ssa173 i128, 0x10, 0x4
|
||||
* %ssa175 i128 = VAdd %ssa172 i128, %ssa173 i128, 0x10, 0x4
|
||||
* (%%ssa176) StoreContext %ssa175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
@@ -698,6 +765,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
|
||||
// XXX: We don't do cross-block optimizations yet
|
||||
//CalculateControlFlowInfo(IREmit);
|
||||
bool Changed = false;
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
@@ -154,6 +155,8 @@ struct Info {
|
||||
*
|
||||
*/
|
||||
bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
std::unordered_map<OrderedNode*, Info> InfoMap;
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
@@ -52,6 +53,8 @@ IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
}
|
||||
|
||||
bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRCompaction");
|
||||
|
||||
LocalBuilder.ReownOrClaimBuffer();
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
@@ -32,6 +33,8 @@ IRValidation::~IRValidation() {
|
||||
}
|
||||
|
||||
bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRValidation");
|
||||
|
||||
bool HadError = false;
|
||||
bool HadWarning = false;
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -53,6 +54,8 @@ bool LongDivideEliminationPass::IsSextOp(IREmitter *IREmit, OrderedNodeWrapper L
|
||||
}
|
||||
|
||||
bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::LDE");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
@@ -24,6 +25,8 @@ public:
|
||||
};
|
||||
|
||||
bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::PHIValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <deque>
|
||||
@@ -191,6 +191,8 @@ private:
|
||||
bool RAValidation::Run(IREmitter *IREmit) {
|
||||
if (!Manager->HasPass("RA")) return false;
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
BlockExitState.clear();
|
||||
// BlocksToVisit will already be empty
|
||||
|
||||
+4
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <array>
|
||||
@@ -32,6 +34,8 @@ public:
|
||||
*
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
std::array<OrderedNode*, 32> LastValidFlagStores{};
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -15,6 +15,8 @@ $end_info$
|
||||
#include <FEXCore/Utils/BucketList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -1527,6 +1529,7 @@ namespace {
|
||||
}
|
||||
|
||||
bool ConstrainedRAPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RA");
|
||||
bool Changed = false;
|
||||
|
||||
auto IR = IREmit->ViewIR();
|
||||
|
||||
-128
@@ -1,128 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Replaces Load/StoreContext with Load/StoreReg for SRA regs
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class StaticRegisterAllocationPass final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit StaticRegisterAllocationPass(bool SupportsAVX_) : SupportsAVX{SupportsAVX_} {}
|
||||
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
|
||||
private:
|
||||
bool SupportsAVX;
|
||||
|
||||
bool IsStaticAllocGpr(uint32_t Offset, RegisterClassType Class) const {
|
||||
const auto begin = offsetof(Core::CPUState, gregs[0]);
|
||||
const auto end = offsetof(Core::CPUState, gregs[16]);
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto reg = (Offset - begin) / Core::CPUState::GPR_REG_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(Class.Val == IR::GPRClass.Val, "unexpected Class {}", Class);
|
||||
|
||||
// 0..15 -> 16 in total
|
||||
return reg < Core::CPUState::NUM_GPRS;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool IsStaticAllocFpr(uint32_t Offset, RegisterClassType Class, bool AllowGpr) const {
|
||||
const auto [begin, end] = [this]() -> std::pair<ptrdiff_t, ptrdiff_t> {
|
||||
if (SupportsAVX) {
|
||||
return {
|
||||
offsetof(Core::CPUState, xmm.avx.data[0][0]),
|
||||
offsetof(Core::CPUState, xmm.avx.data[16][0]),
|
||||
};
|
||||
} else {
|
||||
return {
|
||||
offsetof(Core::CPUState, xmm.sse.data[0][0]),
|
||||
offsetof(Core::CPUState, xmm.sse.data[16][0]),
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto size = SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto reg = (Offset - begin) / size;
|
||||
LOGMAN_THROW_AA_FMT(Class.Val == IR::FPRClass.Val || (AllowGpr && Class.Val == IR::GPRClass.Val), "unexpected Class {}, AllowGpr {}", Class, AllowGpr);
|
||||
|
||||
// 0..15 -> 16 in total
|
||||
return reg < Core::CPUState::NUM_XMMS;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This pass replaces Load/Store Context with Load/Store Register for Statically Mapped registers. It also does some validation.
|
||||
*
|
||||
*/
|
||||
bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
|
||||
if (IsStaticAllocGpr(Op->Offset, Op->Class) || IsStaticAllocFpr(Op->Offset, Op->Class, true)) {
|
||||
auto GeneralClass = Op->Class;
|
||||
if (IsStaticAllocFpr(Op->Offset, GeneralClass, true) && GeneralClass == GPRClass) {
|
||||
GeneralClass = FPRClass;
|
||||
}
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
OrderedNode *sraReg = IREmit->_LoadRegister(false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
if (GeneralClass != Op->Class) {
|
||||
sraReg = IREmit->_VExtractToGPR(Op->Header.Size, Op->Header.Size, sraReg, 0);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, sraReg);
|
||||
}
|
||||
} if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
|
||||
if (IsStaticAllocGpr(Op->Offset, Op->Class) || IsStaticAllocFpr(Op->Offset, Op->Class, true)) {
|
||||
auto val = IREmit->UnwrapNode(Op->Value);
|
||||
|
||||
auto GeneralClass = Op->Class;
|
||||
if (IsStaticAllocFpr(Op->Offset, GeneralClass, true) && GeneralClass == GPRClass) {
|
||||
val = IREmit->_VCastFromGPR(Op->Header.Size, Op->Header.Size, val);
|
||||
GeneralClass = FPRClass;
|
||||
}
|
||||
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
IREmit->_StoreRegister(val, false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX) {
|
||||
return std::make_unique<StaticRegisterAllocationPass>(SupportsAVX);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -23,6 +24,8 @@ public:
|
||||
};
|
||||
|
||||
bool SyscallOptimization::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
@@ -36,6 +37,8 @@ public:
|
||||
};
|
||||
|
||||
bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ValueDominanceValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
+38
-7
@@ -45,6 +45,8 @@ namespace FEXCore::Allocator {
|
||||
FREE_Hook free {::free};
|
||||
#endif
|
||||
|
||||
uint64_t HostVASize{};
|
||||
|
||||
using GLIBC_MALLOC_Hook = void*(*)(size_t, const void *caller);
|
||||
using GLIBC_REALLOC_Hook = void*(*)(void*, size_t, const void *caller);
|
||||
using GLIBC_FREE_Hook = void(*)(void*, const void *caller);
|
||||
@@ -97,6 +99,10 @@ namespace FEXCore::Allocator {
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
|
||||
if (HostVASize) {
|
||||
return HostVASize;
|
||||
}
|
||||
|
||||
static constexpr std::array<uintptr_t, 7> TLBSizes = {
|
||||
57,
|
||||
52,
|
||||
@@ -127,6 +133,7 @@ namespace FEXCore::Allocator {
|
||||
};
|
||||
|
||||
if (Find(Size)) {
|
||||
HostVASize = Bits;
|
||||
return Bits;
|
||||
}
|
||||
}
|
||||
@@ -138,8 +145,10 @@ namespace FEXCore::Allocator {
|
||||
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
|
||||
|
||||
std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
void * const StackLocation = alloca(0);
|
||||
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(StackLocation);
|
||||
std::vector<MemoryRegion> Regions;
|
||||
|
||||
|
||||
int MapsFD = open("/proc/self/maps", O_RDONLY);
|
||||
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
|
||||
|
||||
@@ -148,6 +157,8 @@ namespace FEXCore::Allocator {
|
||||
uintptr_t RegionBegin = 0;
|
||||
uintptr_t RegionEnd = 0;
|
||||
|
||||
uintptr_t PreviousMapEnd = 0;
|
||||
|
||||
char Buffer[2048];
|
||||
const char *Cursor;
|
||||
ssize_t Remaining = 0;
|
||||
@@ -155,7 +166,7 @@ namespace FEXCore::Allocator {
|
||||
for(;;) {
|
||||
|
||||
if (Remaining == 0) {
|
||||
do {
|
||||
do {
|
||||
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
|
||||
} while ( Remaining == -1 && errno == EAGAIN);
|
||||
|
||||
@@ -165,8 +176,8 @@ namespace FEXCore::Allocator {
|
||||
if (Remaining == 0 && State == ParseBegin) {
|
||||
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = End;
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = End;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
@@ -202,9 +213,12 @@ namespace FEXCore::Allocator {
|
||||
if (c == '-') {
|
||||
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
// Store the location we are going to map.
|
||||
PreviousMapEnd = MapEnd;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
if (MapEnd > MapBegin) {
|
||||
@@ -218,6 +232,7 @@ namespace FEXCore::Allocator {
|
||||
|
||||
Regions.push_back({(void*)MapBegin, MapSize});
|
||||
}
|
||||
|
||||
RegionBegin = 0;
|
||||
RegionEnd = 0;
|
||||
State = ParseEnd;
|
||||
@@ -233,6 +248,22 @@ namespace FEXCore::Allocator {
|
||||
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
State = ScanEnd;
|
||||
|
||||
// If the previous map's ending and the region we just parsed overlap the stack then we need to save the stack mapping.
|
||||
// Otherwise we will have severely limited stack size which crashes quickly.
|
||||
if (PreviousMapEnd <= StackLocation_u64 && RegionEnd > StackLocation_u64) {
|
||||
auto BelowStackRegion = Regions.back();
|
||||
LOGMAN_THROW_AA_FMT(reinterpret_cast<uint64_t>(BelowStackRegion.Ptr) + BelowStackRegion.Size == PreviousMapEnd,
|
||||
"This needs to match");
|
||||
|
||||
// Allocate the region under the stack as READ | WRITE so the stack can still grow
|
||||
auto Alloc = mmap(BelowStackRegion.Ptr, BelowStackRegion.Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", BelowStackRegion.Ptr, BelowStackRegion.Size);
|
||||
LogMan::Throw::AFmt(Alloc == BelowStackRegion.Ptr, "mmap({},{:x}) returned {} instead of {:x}", Alloc, BelowStackRegion.Ptr);
|
||||
|
||||
Regions.pop_back();
|
||||
}
|
||||
continue;
|
||||
} else {
|
||||
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
|
||||
|
||||
Loaded 100 of 507 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user