mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 03:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cac3767c20 | ||
|
|
e82d2c70c3 | ||
|
|
9716fc73d3 | ||
|
|
aaa8eef9f1 | ||
|
|
c036868938 | ||
|
|
2726f35a10 | ||
|
|
407dc5d4dd | ||
|
|
7dda75646d | ||
|
|
b89f5b8a03 | ||
|
|
16869d8954 | ||
|
|
e353ae8408 | ||
|
|
aa3c963df1 | ||
|
|
0c8cfa7bb2 | ||
|
|
dd837fa693 | ||
|
|
eeff198ff1 | ||
|
|
0cd6371c1b | ||
|
|
5aff16f72b | ||
|
|
2c7896bce4 | ||
|
|
6dd72a09e5 | ||
|
|
4e6a4d9b69 | ||
|
|
44484a7b05 | ||
|
|
6ac7ba388f | ||
|
|
36d0a67070 | ||
|
|
0650dd1992 | ||
|
|
e2d58809ed | ||
|
|
967a74cda9 | ||
|
|
6cc6181261 | ||
|
|
f175b525f4 | ||
|
|
319f1e66cb | ||
|
|
d16969bdce | ||
|
|
c17858bfeb | ||
|
|
ec24fc3d5d | ||
|
|
8819fa88d5 | ||
|
|
5e9c2110db | ||
|
|
5aaa18e7a2 | ||
|
|
8a551b9e64 | ||
|
|
aa548bd19c | ||
|
|
9565f16d84 | ||
|
|
eb76dbdf4d | ||
|
|
967c04e252 | ||
|
|
49de1fac59 | ||
|
|
3c205eb35e | ||
|
|
682b8ef705 | ||
|
|
8c47a40625 | ||
|
|
ee5c6af868 | ||
|
|
1670c89b5e | ||
|
|
e9d34c7ded | ||
|
|
c2f2c58367 | ||
|
|
38acd9e18c | ||
|
|
21604708ca | ||
|
|
5fec6faeb6 | ||
|
|
b659701cef | ||
|
|
e902ad5278 | ||
|
|
0f54369f9b | ||
|
|
f640dcc7a4 | ||
|
|
b59da0e049 | ||
|
|
4ae3aef502 | ||
|
|
926fa3c24c | ||
|
|
8c1740f592 | ||
|
|
611aa0a5b9 | ||
|
|
0b8b5108b9 | ||
|
|
e409a0afec | ||
|
|
24211f8523 | ||
|
|
94462e4dd0 | ||
|
|
aa44668a41 | ||
|
|
33a675624c | ||
|
|
5745b419d9 | ||
|
|
9100235041 | ||
|
|
4d62e75d5a | ||
|
|
b6d3df0e54 | ||
|
|
d5d1230422 | ||
|
|
f35a212f06 | ||
|
|
f49d82deb3 | ||
|
|
77b801b8ae | ||
|
|
163a3fc899 | ||
|
|
9b627e8743 | ||
|
|
9d6865f62d | ||
|
|
d547f2b8f2 | ||
|
|
b5c9eb3463 | ||
|
|
00a3793814 | ||
|
|
4f8d7cb89c | ||
|
|
9310ac59f4 | ||
|
|
9e4de3bfe8 | ||
|
|
598b99fe58 | ||
|
|
9dda76ebe6 | ||
|
|
f5def7ae1c | ||
|
|
58bcf691d6 | ||
|
|
33186b803d | ||
|
|
bb1d7d0750 | ||
|
|
4897c1e80c | ||
|
|
e7605c94dd | ||
|
|
ab5a10374b | ||
|
|
dc2a26582f | ||
|
|
e6512fbe4d | ||
|
|
dbca441573 | ||
|
|
31adc953b7 | ||
|
|
0dd7f5e2f7 | ||
|
|
36556c4705 | ||
|
|
e29fac3f25 | ||
|
|
c984bdb42a | ||
|
|
2645b374a7 | ||
|
|
10192eebe5 | ||
|
|
0e57cbf5b9 | ||
|
|
c7413d96ad | ||
|
|
f3ba8cb33c | ||
|
|
e714933b10 | ||
|
|
9a0f83cbf6 | ||
|
|
c4d30e8437 | ||
|
|
7b30df8a56 | ||
|
|
86aa459dd1 | ||
|
|
6c9a47fc71 | ||
|
|
7a72cf631b | ||
|
|
3d98eef7b8 | ||
|
|
f2011b0b79 | ||
|
|
c0b8a0d9c3 | ||
|
|
5bd0afa307 | ||
|
|
20fb0da7d9 | ||
|
|
35cb1f710c | ||
|
|
7c7b99f9c5 | ||
|
|
09879bd962 | ||
|
|
c8f3fe3788 | ||
|
|
88e3164b55 | ||
|
|
bebd7400c0 | ||
|
|
e190d029dc | ||
|
|
37a70e2ec6 | ||
|
|
6ff073ac11 | ||
|
|
b6e4c47abc | ||
|
|
daf8b409f3 | ||
|
|
0a35029d9a | ||
|
|
e8b51ac0a4 | ||
|
|
f20e626fda | ||
|
|
2672e06aa5 | ||
|
|
4a2b4be59d | ||
|
|
05be9445cd | ||
|
|
4cb4c0e277 | ||
|
|
0135e2d78c | ||
|
|
1266a5a5ce | ||
|
|
7255850e9d | ||
|
|
002a8bebd8 | ||
|
|
452c9fd19b | ||
|
|
0b0e15c9b4 | ||
|
|
ae9db336e7 | ||
|
|
360ea538b7 | ||
|
|
f99900bdc5 | ||
|
|
34f2bce203 | ||
|
|
a05d6ae082 | ||
|
|
6a4eb434f7 | ||
|
|
0c3eb41f57 | ||
|
|
e4bfbda008 | ||
|
|
b4c797c199 | ||
|
|
ebae1bf71d | ||
|
|
fa293e5e8f | ||
|
|
720c89e648 | ||
|
|
abc5f5abd4 | ||
|
|
22ab16b217 | ||
|
|
b80f05dd86 | ||
|
|
1642a0f4d9 | ||
|
|
96e990dade | ||
|
|
3423a12ab6 | ||
|
|
c4a0fb6609 | ||
|
|
8cf23eba9c | ||
|
|
fbe817f1c4 | ||
|
|
3b61394548 | ||
|
|
8ed6b36181 | ||
|
|
9141322666 | ||
|
|
2d91c5441e | ||
|
|
2ff1589c35 | ||
|
|
a308b9edbe | ||
|
|
f6a8e595df | ||
|
|
eaaf62e4b7 | ||
|
|
dc5fc57e19 | ||
|
|
c3c0bbb060 | ||
|
|
98a91d242d | ||
|
|
8da6b1e729 | ||
|
|
d260bce31e | ||
|
|
a94aaca0ec | ||
|
|
2ac3ca3ead | ||
|
|
c559445b62 | ||
|
|
8ea4e4f663 | ||
|
|
2da700fe82 | ||
|
|
0a32ba9b29 | ||
|
|
f51832ca6b | ||
|
|
5f9a30af7a | ||
|
|
453ef0b94c | ||
|
|
4779e64cf5 | ||
|
|
e6025cc087 | ||
|
|
cd4c224dc0 | ||
|
|
07f117b957 | ||
|
|
36e4f9a070 | ||
|
|
67c751d3e5 | ||
|
|
1c59bfeb9f | ||
|
|
2b0041a291 | ||
|
|
39cc1e5116 | ||
|
|
f46856c090 | ||
|
|
b7859052da | ||
|
|
3b30d8103f | ||
|
|
d19fc36f8f | ||
|
|
5fbf76ee9e | ||
|
|
5f79761fe4 | ||
|
|
621af9e7e8 | ||
|
|
f4d107c8f0 | ||
|
|
59643db331 | ||
|
|
46f36dc89d | ||
|
|
384882744c | ||
|
|
b36d1f7e7b | ||
|
|
dd7c66db2e | ||
|
|
47ac9edd7a | ||
|
|
6e6d640fec | ||
|
|
eb88366614 | ||
|
|
50d6ddd591 | ||
|
|
249de7c758 | ||
|
|
c350888d70 | ||
|
|
f23bef1653 | ||
|
|
d17f33a922 | ||
|
|
50a3ca0d6d | ||
|
|
8745455a5b | ||
|
|
e9ab514962 | ||
|
|
4eb0948451 | ||
|
|
8d9f19bd73 | ||
|
|
9fe7bc5818 | ||
|
|
5a590a9b11 | ||
|
|
a4acd64246 | ||
|
|
304b5de1af | ||
|
|
ca2fe0b301 | ||
|
|
bbef4d762c | ||
|
|
f77841d784 | ||
|
|
219a4777c6 | ||
|
|
29f0e9b4d6 | ||
|
|
b75efc42bf | ||
|
|
90dbd4766d | ||
|
|
f7b911ca43 | ||
|
|
6de0333708 | ||
|
|
ab5d3ab22b | ||
|
|
355c3428c0 | ||
|
|
c76b7bfe8d | ||
|
|
64c5362580 | ||
|
|
77469de247 | ||
|
|
48fd827004 | ||
|
|
f09d511ac8 | ||
|
|
d5db8948db | ||
|
|
5013b8a0db | ||
|
|
ac65deed6c | ||
|
|
d1d4d2d876 | ||
|
|
e234e118e0 | ||
|
|
1cc54312fa | ||
|
|
de9ab6a023 | ||
|
|
463a0b5ed4 | ||
|
|
7b0365b377 | ||
|
|
803501526e | ||
|
|
a63a3a47e3 | ||
|
|
a66fac614b | ||
|
|
6d4693cbc1 | ||
|
|
46a2a0608b | ||
|
|
49b40b71a6 | ||
|
|
ca8f347902 | ||
|
|
e415b9443f | ||
|
|
fd4f6b8020 | ||
|
|
7a9eb01573 | ||
|
|
b0474816c7 | ||
|
|
941b065da6 | ||
|
|
ccba993ca9 | ||
|
|
e664f61da8 | ||
|
|
c74df6a59b | ||
|
|
fc76d4e8c3 | ||
|
|
099211f39c | ||
|
|
b327d7ae19 | ||
|
|
0e6a22febe | ||
|
|
b368223d50 | ||
|
|
8fe1e9562d | ||
|
|
ae545b8cf3 | ||
|
|
ac32876e4e | ||
|
|
9336e35052 | ||
|
|
0754affb98 | ||
|
|
c413d7950b | ||
|
|
f8eaf9c14f | ||
|
|
fc677eaabf | ||
|
|
114112a716 | ||
|
|
9056d9b9de | ||
|
|
74e95df661 | ||
|
|
d9544e7e02 | ||
|
|
62e1767ee0 | ||
|
|
4baeffe84f | ||
|
|
b4a67a6178 | ||
|
|
8296bfc7de | ||
|
|
f588304b12 | ||
|
|
c748dbf0e3 | ||
|
|
06497fdfad | ||
|
|
e2a7fef742 | ||
|
|
17692d6eef | ||
|
|
3020a0db2b | ||
|
|
92ddc0041b | ||
|
|
8a5388f514 | ||
|
|
a444db7dac | ||
|
|
74da2eb5fc | ||
|
|
539d5ed26f | ||
|
|
cf7ee98831 | ||
|
|
5529948479 | ||
|
|
91b6aeffe1 | ||
|
|
3a79b61f58 | ||
|
|
e92b24302f | ||
|
|
cc9ccf881b | ||
|
|
72d74ae482 | ||
|
|
e88c57bec7 | ||
|
|
ac814ac015 | ||
|
|
90f7cc925d | ||
|
|
2039950762 | ||
|
|
b526c60c73 | ||
|
|
1bfdd031d3 | ||
|
|
49ee8ba3be | ||
|
|
335cd9180e | ||
|
|
ca9a94d572 | ||
|
|
03832b2523 | ||
|
|
812224a0ef | ||
|
|
42f2851575 | ||
|
|
9c7f44f8d4 | ||
|
|
5117ba351e | ||
|
|
441fdb689d | ||
|
|
e786dfc998 | ||
|
|
a1d3183c14 | ||
|
|
c5ef0910c5 | ||
|
|
affa1d0efc | ||
|
|
a83dc27a42 | ||
|
|
129ec63e92 | ||
|
|
3459369c6e | ||
|
|
7a490a3811 | ||
|
|
ee17fe239a | ||
|
|
0416950aaa | ||
|
|
6e46383cdb | ||
|
|
faf1b85904 | ||
|
|
bec5f4fe2a | ||
|
|
fe6dbbfa63 | ||
|
|
b19440c78f | ||
|
|
abf9700475 | ||
|
|
8d6b454455 | ||
|
|
205ec3e14d | ||
|
|
f897579593 | ||
|
|
2478abba29 | ||
|
|
a4db585664 | ||
|
|
8bf4a124c8 | ||
|
|
13c3b65732 | ||
|
|
7e6ba184f9 | ||
|
|
f6cb914a2a | ||
|
|
6c92f94ee8 | ||
|
|
bebf4209d6 | ||
|
|
c82a683987 | ||
|
|
886d40ccfe | ||
|
|
8f33e56e21 | ||
|
|
b1634680fe | ||
|
|
54d332935e | ||
|
|
2829ad56a1 | ||
|
|
fbf62f1296 | ||
|
|
e6aa268093 | ||
|
|
cc589ba7e6 | ||
|
|
f009a00986 | ||
|
|
8617150a42 | ||
|
|
ef823ce82b | ||
|
|
fc2917117e | ||
|
|
68abe400a7 | ||
|
|
df86d80a85 | ||
|
|
58cff72d38 | ||
|
|
fd1f5643c7 | ||
|
|
0ea29dfbdf | ||
|
|
f4fca4482f | ||
|
|
86d1ce7f00 | ||
|
|
15fef1794c | ||
|
|
8b5d9d8fdf | ||
|
|
0ec724cf1f | ||
|
|
0dc0117e86 | ||
|
|
1d00ad6030 | ||
|
|
23a076c313 | ||
|
|
57eacab654 | ||
|
|
b4093a8888 | ||
|
|
7c5a9b5d6a | ||
|
|
12b3c82d83 | ||
|
|
66520bce0a | ||
|
|
25cc2bdcb8 | ||
|
|
f98b18800c | ||
|
|
0dd687a7a1 | ||
|
|
1aff3acbb9 | ||
|
|
a6ab2ca30d | ||
|
|
ca43e2a61c | ||
|
|
4f404160d0 | ||
|
|
5edc69b692 | ||
|
|
b6cb897896 | ||
|
|
47b3637452 | ||
|
|
5b65f30c8f | ||
|
|
23572539f8 | ||
|
|
7176c717e5 | ||
|
|
3a05b760d4 | ||
|
|
1e0273d570 | ||
|
|
8232be6302 | ||
|
|
43cd897cf4 | ||
|
|
e593807856 | ||
|
|
ce8e6e5c0c | ||
|
|
5b9a7c7845 | ||
|
|
0e5f5a2db9 | ||
|
|
50a9cea16e | ||
|
|
6aa95d82f2 | ||
|
|
7c34f449d1 | ||
|
|
e9435203a0 | ||
|
|
8d9ba0e102 | ||
|
|
3fcfd2ccde | ||
|
|
aa4205c2e8 | ||
|
|
5d5b411611 | ||
|
|
5f5a06fb3b | ||
|
|
3077addcf8 | ||
|
|
8b6fe0c4ff | ||
|
|
29fd62e9ba | ||
|
|
2631b113da | ||
|
|
da152031b3 | ||
|
|
be9c5678c0 | ||
|
|
fdd370dd1a | ||
|
|
6951284924 | ||
|
|
b03c613fbf | ||
|
|
b893bdb8df | ||
|
|
db45f6eec8 | ||
|
|
d53e689e22 | ||
|
|
f55378257a | ||
|
|
6a7914ac56 | ||
|
|
d935d25f0c | ||
|
|
84cf1d2fd2 | ||
|
|
ed0c045c17 | ||
|
|
d585063e60 | ||
|
|
3077fec9ab | ||
|
|
9500842efc | ||
|
|
a6f9c51317 | ||
|
|
cfa2ad8423 | ||
|
|
19010491da | ||
|
|
5d613e8716 | ||
|
|
894aaa980f | ||
|
|
877b2f4fef | ||
|
|
2a170cfdec | ||
|
|
a299d6b1a5 | ||
|
|
ffb85e6305 | ||
|
|
d7a20fa28f | ||
|
|
5ac7d5dfcd | ||
|
|
75644b33df | ||
|
|
4c4c6e7807 | ||
|
|
a8c9c71ce3 | ||
|
|
8a4bd5f22c | ||
|
|
138a36c69a | ||
|
|
3400ca5d42 | ||
|
|
5d8164da4e | ||
|
|
8b89b30a6f | ||
|
|
e66d7cfd6c | ||
|
|
63afa29dae | ||
|
|
926a9b40e2 | ||
|
|
9bb43264b5 | ||
|
|
32d6daf558 | ||
|
|
bebcb73c68 | ||
|
|
d0e040514f | ||
|
|
850c027d52 | ||
|
|
5df90563e2 | ||
|
|
7a489d18c6 | ||
|
|
1db092e96f | ||
|
|
65a4de221b | ||
|
|
9c8df79dfb | ||
|
|
689b461d7b | ||
|
|
92c951c81f | ||
|
|
9c8438f264 | ||
|
|
caf7ad53e6 | ||
|
|
47f0fec2f2 | ||
|
|
eadb502059 | ||
|
|
8e1695afa3 | ||
|
|
a7138f26b6 | ||
|
|
f2a9ce9d4c | ||
|
|
84e4960e52 | ||
|
|
99afd876ba | ||
|
|
1caa31c5cb | ||
|
|
d2c82ba707 | ||
|
|
9067f3513a | ||
|
|
409d691877 | ||
|
|
1762ff0393 | ||
|
|
4d26178f25 | ||
|
|
c6582a1ce5 | ||
|
|
a51be56c9e | ||
|
|
e5cb583cc0 | ||
|
|
3ef83a9e23 | ||
|
|
7de6a5cb07 | ||
|
|
e06b3a4186 | ||
|
|
c495f82f4f | ||
|
|
49e4426ed5 | ||
|
|
c4e2436885 | ||
|
|
f078b25c7d | ||
|
|
e5149fba57 | ||
|
|
96055cbde7 | ||
|
|
4abac0cac7 | ||
|
|
1e1bcc4af2 | ||
|
|
f1d7879365 | ||
|
|
0ea3de95e0 | ||
|
|
33cef7c8cd | ||
|
|
b0fd220f3e | ||
|
|
1e3539537d | ||
|
|
1b13a3dd4e | ||
|
|
d1ec242e4f | ||
|
|
21f9841e9d | ||
|
|
ea6c05a46a | ||
|
|
226f5e2f23 | ||
|
|
a82fcdecd7 | ||
|
|
27acbe305d | ||
|
|
cd0739a534 | ||
|
|
f1055d0713 | ||
|
|
7875b20594 | ||
|
|
25679cd319 | ||
|
|
fa45ec32db | ||
|
|
40d12c4d0e | ||
|
|
5e706dff64 | ||
|
|
a4860d6f87 | ||
|
|
ef4c4f6e9b | ||
|
|
abbd65549a | ||
|
|
00ef1baead | ||
|
|
75687070a5 | ||
|
|
9b1496c327 | ||
|
|
27cce9d9b7 | ||
|
|
08b66bf827 | ||
|
|
0aa5807482 | ||
|
|
7a2d8c5c01 | ||
|
|
1b8b1a24b0 | ||
|
|
86eb2b13b9 | ||
|
|
c634c53434 | ||
|
|
2eb7a9ff28 | ||
|
|
933c65d805 | ||
|
|
df0ecad15b | ||
|
|
ce88f5f948 | ||
|
|
27cd399d6f | ||
|
|
7b70925acf | ||
|
|
2a3e0b93e8 | ||
|
|
2b26d0fff5 | ||
|
|
129f676610 | ||
|
|
2dc92c122e | ||
|
|
0db17bd58a | ||
|
|
aac16493f1 | ||
|
|
86e5e1a15e | ||
|
|
aa5d2ff31c | ||
|
|
30dd5f5c75 | ||
|
|
f5bc06476a | ||
|
|
c70e44cf05 | ||
|
|
ede13e37d9 | ||
|
|
a7dada457a | ||
|
|
6367554d30 | ||
|
|
81969a684e | ||
|
|
ee339b5960 | ||
|
|
881c940693 | ||
|
|
900c62fa7b | ||
|
|
200c6c054f | ||
|
|
bc0927b7b1 | ||
|
|
231c28395c | ||
|
|
64a45c0d29 | ||
|
|
cab02be637 | ||
|
|
74f341bc0e | ||
|
|
feaa1af1a8 | ||
|
|
d59d040b4e | ||
|
|
a9c26cbf71 | ||
|
|
f4b5c4e69a | ||
|
|
b8cac9f7d5 | ||
|
|
13974df204 | ||
|
|
fa6fe9bf06 | ||
|
|
746be0824e | ||
|
|
93120cabbb | ||
|
|
04ae05f4ce | ||
|
|
83773dddc7 | ||
|
|
9813553f02 | ||
|
|
924723d433 | ||
|
|
33558e63c4 | ||
|
|
97c229d5eb | ||
|
|
16b007df33 | ||
|
|
3cbc421c7e | ||
|
|
890e5e1f0f | ||
|
|
6b98454f03 | ||
|
|
b05329c8df | ||
|
|
6f43c8ffac | ||
|
|
ffac98051f | ||
|
|
40812efaae | ||
|
|
3429321d59 | ||
|
|
6cddd6cbe7 | ||
|
|
23d07d7d0c | ||
|
|
1351575713 | ||
|
|
8aa7d1a278 | ||
|
|
3383786205 | ||
|
|
91f4c54768 | ||
|
|
d9c779289c | ||
|
|
5631ff4fd5 | ||
|
|
34301319bf | ||
|
|
8eac3198b6 | ||
|
|
832edd4da3 | ||
|
|
5823e74bcd | ||
|
|
a4545f493e | ||
|
|
3b8cd44ca4 | ||
|
|
f138d7d9b8 | ||
|
|
3bb9d44bf5 | ||
|
|
06e6a1b19e | ||
|
|
256d166126 | ||
|
|
630285c589 | ||
|
|
9621eca677 | ||
|
|
7eee50d929 | ||
|
|
1dc22e33ae | ||
|
|
0ef8aaebeb | ||
|
|
d17c427e47 | ||
|
|
f73fb62c6e | ||
|
|
5976b712ca | ||
|
|
63f5e64adb | ||
|
|
34ee3bb8aa | ||
|
|
79c745929a | ||
|
|
8330cc6876 | ||
|
|
16e6163677 | ||
|
|
bbbc0dc9dc | ||
|
|
ea4004ec9d | ||
|
|
a469047e7a | ||
|
|
2cc0a867a9 |
No files matched your search
+1
-1
@@ -4,7 +4,7 @@ compile_commands.json
|
||||
vim_rc
|
||||
Config.json
|
||||
|
||||
[Bb]uild*/
|
||||
[Bb]uild*
|
||||
[Bb]in/
|
||||
out/
|
||||
.vscode/
|
||||
|
||||
@@ -5,15 +5,6 @@
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/imgui"]
|
||||
path = External/imgui
|
||||
url = https://github.com/Sonicadvance1/imgui.git
|
||||
[submodule "External/json-maker"]
|
||||
path = External/json-maker
|
||||
url = https://github.com/Sonicadvance1/json-maker.git
|
||||
[submodule "External/tiny-json"]
|
||||
path = External/tiny-json
|
||||
url = https://github.com/Sonicadvance1/tiny-json.git
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
|
||||
+33
-24
@@ -7,7 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -28,6 +28,7 @@ option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
@@ -48,7 +49,7 @@ endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
set (CLANG_MINIMUM_VERSION 12.0)
|
||||
set (CLANG_MINIMUM_VERSION 13.0)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
|
||||
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
endif()
|
||||
@@ -231,7 +232,6 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
# The glibc jemalloc subproject which hooks the glibc allocator.
|
||||
# Required for thunks to work.
|
||||
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
|
||||
add_definitions(-DENABLE_JEMALLOC_GLIBC=1)
|
||||
add_subdirectory(External/jemalloc_glibc/)
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
@@ -243,9 +243,7 @@ endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
# The jemalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
" jemalloc disabled!\n"
|
||||
@@ -272,8 +270,10 @@ if (BUILD_TESTS)
|
||||
set(COMPILE_VIXL_DISASSEMBLER TRUE)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
if (COMPILE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
@@ -283,31 +283,38 @@ endif()
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
|
||||
pkg_search_module(xxhash IMPORTED_TARGET xxhash libxxhash)
|
||||
if (TARGET PkgConfig::xxhash AND NOT CMAKE_CROSSCOMPILING)
|
||||
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
|
||||
else()
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
endif()
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(External/Catch2/)
|
||||
find_package(Catch2 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
endif()
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
|
||||
add_subdirectory(External/json-maker/)
|
||||
include_directories(External/json-maker/)
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
include_directories(External/tiny-json/)
|
||||
@@ -423,8 +430,10 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
# This is a reference AArch64 cross compile script
|
||||
# Pass in to cmake when building:
|
||||
# eg: cmake -DCMAKE_TOOLCHAIN_FILE=../CMakeToolchains/AArch64.cmake ..
|
||||
if (NOT DEFINED ENV{SYSROOT})
|
||||
message(FATAL_ERROR "Need to have SYSROOT environment variable set")
|
||||
endif()
|
||||
|
||||
set(CMAKE_SYSTEM_NAME Linux)
|
||||
set(CMAKE_SYSTEM_PROCESSOR aarch64)
|
||||
set(CMAKE_CROSSCOMPILING TRUE)
|
||||
|
||||
# Target triple needs to match the binutils exactly
|
||||
set(TARGET_TRIPLE aarch64-linux-gnu)
|
||||
set(CMAKE_C_COMPILER "clang")
|
||||
set(CMAKE_CXX_COMPILER "clang++")
|
||||
set(CMAKE_C_COMPILER_AR "llvm-ar")
|
||||
set(CMAKE_CXX_COMPILER_AR "llvm-ar")
|
||||
set(CMAKE_C_COMPILER_RANLIB "llvm-ranlib")
|
||||
set(CMAKE_CXX_COMPILER_RANLIB "llvm-ranlib")
|
||||
set(CMAKE_LINKER "ld.lld")
|
||||
|
||||
set(CMAKE_C_COMPILER_TARGET ${TARGET_TRIPLE})
|
||||
set(CMAKE_CXX_COMPILER_TARGET ${TARGET_TRIPLE})
|
||||
|
||||
# Set the environment variable SYSROOT to the aarch64 rootfs
|
||||
set(CMAKE_FIND_ROOT_PATH "$ENV{SYSROOT}")
|
||||
set(CMAKE_SYSROOT "$ENV{SYSROOT}")
|
||||
|
||||
list(APPEND CMAKE_PREFIX_PATH "$ENV{SYSROOT}/usr/lib/${TARGET_TRIPLE}/cmake/")
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER)
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY)
|
||||
@@ -301,8 +301,9 @@ public:
|
||||
xbfiz_helper(true, s, rd, rn, lsb, width);
|
||||
}
|
||||
void asr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
sbfm(s, rd, rn, shift, RegSize_m1);
|
||||
}
|
||||
|
||||
void uxtb(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
@@ -325,14 +326,14 @@ public:
|
||||
}
|
||||
|
||||
void lsl(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
ubfm(s, rd, rn, (RegSize - shift) % RegSize, RegSize - shift - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
ubfm(s, rd, rn, (RegSizeInBits(s) - shift) & RegSize_m1, RegSize_m1 - shift);
|
||||
}
|
||||
void lsr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsr a region larger than the register");
|
||||
ubfm(s, rd, rn, shift, RegSize - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
ubfm(s, rd, rn, shift, RegSize_m1);
|
||||
}
|
||||
void ubfx(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
LOGMAN_THROW_A_FMT(width > 0, "ubfx needs width > 0");
|
||||
@@ -368,6 +369,7 @@ public:
|
||||
}
|
||||
|
||||
void ror(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t Imm) {
|
||||
Imm &= RegSizeInBits(s) - 1;
|
||||
extr(s, rd, rn, rn, Imm);
|
||||
}
|
||||
|
||||
|
||||
@@ -1070,7 +1070,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10110, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1082,7 +1082,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10110, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1095,7 +1095,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10111, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1107,7 +1107,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10111, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1123,7 +1123,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11000, rd, rn);
|
||||
}
|
||||
@@ -1138,7 +1138,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11001, rd, rn);
|
||||
}
|
||||
@@ -1154,7 +1154,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
@@ -1169,7 +1169,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
@@ -1184,7 +1184,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
@@ -1199,7 +1199,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -1214,7 +1214,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11110, rd, rn);
|
||||
}
|
||||
@@ -1229,7 +1229,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11111, rd, rn);
|
||||
}
|
||||
|
||||
@@ -13,5 +13,14 @@ function(GenBinFmt Name)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
endfunction()
|
||||
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
configure_file(FEX-x86.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf)
|
||||
configure_file(FEX-x86_64.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf)
|
||||
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
|
||||
else()
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
endif()
|
||||
@@ -0,0 +1 @@
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
@@ -6,4 +6,3 @@ mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xf
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
expose_interpreter optional
|
||||
@@ -0,0 +1 @@
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
@@ -6,4 +6,3 @@ mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xf
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
expose_interpreter optional
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 31aa7f634b...29f979ee5a.
Vendored
+1
-1
Submodule External/drm-headers updated: 34a20394f7...8efb6dc03f.
Vendored
+1
-1
Submodule External/fmt updated: f5e54359df...0c9fce2ffe.
Vendored
-1
Submodule External/imgui deleted from 4c986ecb8d.
Vendored
+1
-1
Submodule External/jemalloc updated: 7ae889695b...02ca52b5fe.
Vendored
+1
-1
Submodule External/jemalloc_glibc updated: 888181c5f7...404353974e.
Vendored
-1
Submodule External/json-maker deleted from 8ecb8ecc34.
Vendored
+1
-1
Submodule External/robin-map updated: f1ab690046...d5683d9f18.
Vendored
-1
Submodule External/tiny-json deleted from 9d09127f87.
Vendored
+3
@@ -0,0 +1,3 @@
|
||||
set(NAME tiny-json)
|
||||
set(SRCS tiny-json.c)
|
||||
add_library(${NAME} ${SRCS})
|
||||
Vendored
+21
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2018 Rafa Garcia
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Vendored
+647
@@ -0,0 +1,647 @@
|
||||
|
||||
/*
|
||||
|
||||
<https://github.com/rafagafe/tiny-json>
|
||||
|
||||
Licensed under the MIT License <http://opensource.org/licenses/MIT>.
|
||||
SPDX-License-Identifier: MIT
|
||||
Copyright (c) 2016-2018 Rafa Garcia <rafagarcia77@gmail.com>.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <ctype.h>
|
||||
#include <stddef.h> // For NULL
|
||||
#include "tiny-json.h"
|
||||
|
||||
/** Structure to handle a heap of JSON properties. */
|
||||
typedef struct jsonStaticPool_s {
|
||||
json_t* const mem; /**< Pointer to array of json properties. */
|
||||
unsigned int const qty; /**< Length of the array of json properties. */
|
||||
unsigned int nextFree; /**< The index of the next free json property. */
|
||||
jsonPool_t pool;
|
||||
} jsonStaticPool_t;
|
||||
|
||||
/* Search a property by its name in a JSON object. */
|
||||
json_t const* json_getProperty( json_t const* obj, char const* property ) {
|
||||
json_t const* sibling;
|
||||
for( sibling = obj->u.c.child; sibling; sibling = sibling->sibling )
|
||||
if ( sibling->name && !strcmp( sibling->name, property ) )
|
||||
return sibling;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Search a property by its name in a JSON object and return its value. */
|
||||
char const* json_getPropertyValue( json_t const* obj, char const* property ) {
|
||||
json_t const* field = json_getProperty( obj, property );
|
||||
if ( !field ) return 0;
|
||||
jsonType_t type = json_getType( field );
|
||||
if ( JSON_ARRAY >= type ) return 0;
|
||||
return json_getValue( field );
|
||||
}
|
||||
|
||||
/* Internal prototypes: */
|
||||
static char* goBlank( char* str );
|
||||
static char* goNum( char* str );
|
||||
static json_t* poolInit( jsonPool_t* pool );
|
||||
static json_t* poolAlloc( jsonPool_t* pool );
|
||||
static char* objValue( char* ptr, json_t* obj, jsonPool_t* pool );
|
||||
static char* setToNull( char* ch );
|
||||
static bool isEndOfPrimitive( char ch );
|
||||
|
||||
/* Parse a string to get a json. */
|
||||
json_t const* json_createWithPool( char *str, jsonPool_t *pool ) {
|
||||
char* ptr = goBlank( str );
|
||||
if ( !ptr || *ptr != '{' ) return 0;
|
||||
json_t* obj = pool->init( pool );
|
||||
obj->name = 0;
|
||||
obj->sibling = 0;
|
||||
obj->u.c.child = 0;
|
||||
ptr = objValue( ptr, obj, pool );
|
||||
if ( !ptr ) return 0;
|
||||
return obj;
|
||||
}
|
||||
|
||||
/* Parse a string to get a json. */
|
||||
json_t const* json_create( char* str, json_t mem[], unsigned int qty ) {
|
||||
jsonStaticPool_t spool = {
|
||||
.mem = mem,
|
||||
.qty = qty,
|
||||
.pool = {
|
||||
.init = poolInit,
|
||||
.alloc = poolAlloc
|
||||
}
|
||||
};
|
||||
return json_createWithPool( str, &spool.pool );
|
||||
}
|
||||
|
||||
/** Get a special character with its escape character. Examples:
|
||||
* 'b' -> '\b', 'n' -> '\n', 't' -> '\t'
|
||||
* @param ch The escape character.
|
||||
* @return The character code. */
|
||||
static char getEscape( char ch ) {
|
||||
static struct { char ch; char code; } const pair[] = {
|
||||
{ '\"', '\"' }, { '\\', '\\' },
|
||||
{ '/', '/' }, { 'b', '\b' },
|
||||
{ 'f', '\f' }, { 'n', '\n' },
|
||||
{ 'r', '\r' }, { 't', '\t' },
|
||||
};
|
||||
unsigned int i;
|
||||
for( i = 0; i < sizeof pair / sizeof *pair; ++i )
|
||||
if ( pair[i].ch == ch )
|
||||
return pair[i].code;
|
||||
return '\0';
|
||||
}
|
||||
|
||||
/** Parse 4 characters.
|
||||
* @Param str Pointer to first digit.
|
||||
* @retval '?' If the four characters are hexadecimal digits.
|
||||
* @retcal '\0' In other cases. */
|
||||
static unsigned char getCharFromUnicode( unsigned char const* str ) {
|
||||
unsigned int i;
|
||||
for( i = 0; i < 4; ++i )
|
||||
if ( !isxdigit( str[i] ) )
|
||||
return '\0';
|
||||
return '?';
|
||||
}
|
||||
|
||||
/** Parse a string and replace the scape characters by their meaning characters.
|
||||
* This parser stops when finds the character '\"'. Then replaces '\"' by '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* parseString( char* str ) {
|
||||
unsigned char* head = (unsigned char*)str;
|
||||
unsigned char* tail = (unsigned char*)str;
|
||||
for( ; *head >= ' '; ++head, ++tail ) {
|
||||
if ( *head == '\"' ) {
|
||||
*tail = '\0';
|
||||
return (char*)++head;
|
||||
}
|
||||
if ( *head == '\\' ) {
|
||||
if ( *++head == 'u' ) {
|
||||
char const ch = getCharFromUnicode( ++head );
|
||||
if ( ch == '\0' ) return 0;
|
||||
*tail = ch;
|
||||
head += 3;
|
||||
}
|
||||
else {
|
||||
char const esc = getEscape( *head );
|
||||
if ( esc == '\0' ) return 0;
|
||||
*tail = esc;
|
||||
}
|
||||
}
|
||||
else *tail = *head;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Parse a string to get the name of a property.
|
||||
* @param str Pointer to first character.
|
||||
* @param property The property to assign the name.
|
||||
* @retval Pointer to first of property value. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* propertyName( char* ptr, json_t* property ) {
|
||||
property->name = ++ptr;
|
||||
ptr = parseString( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
ptr = goBlank( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
if ( *ptr++ != ':' ) return 0;
|
||||
return goBlank( ptr );
|
||||
}
|
||||
|
||||
/** Parse a string to get the value of a property when its type is JSON_TEXT.
|
||||
* @param str Pointer to first character ('\"').
|
||||
* @param property The property to assign the name.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* textValue( char* ptr, json_t* property ) {
|
||||
++property->u.value;
|
||||
ptr = parseString( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_TEXT;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Compare two strings until get the null character in the second one.
|
||||
* @param ptr sub string
|
||||
* @param str main string
|
||||
* @retval Pointer to next character.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* checkStr( char* ptr, char const* str ) {
|
||||
while( *str )
|
||||
if ( *ptr++ != *str++ )
|
||||
return 0;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a primitive value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @param value String with the primitive literal.
|
||||
* @param type The code of the type. ( JSON_BOOLEAN or JSON_NULL )
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* primitiveValue( char* ptr, json_t* property, char const* value, jsonType_t type ) {
|
||||
ptr = checkStr( ptr, value );
|
||||
if ( !ptr || !isEndOfPrimitive( *ptr ) ) return 0;
|
||||
ptr = setToNull( ptr );
|
||||
property->type = type;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a true value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* trueValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "true", JSON_BOOLEAN );
|
||||
}
|
||||
|
||||
/** Parser a string to get a false value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* falseValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "false", JSON_BOOLEAN );
|
||||
}
|
||||
|
||||
/** Parser a string to get a null value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* nullValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "null", JSON_NULL );
|
||||
}
|
||||
|
||||
/** Analyze the exponential part of a real number.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non numerical after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* expValue( char* ptr ) {
|
||||
if ( *ptr == '-' || *ptr == '+' ) ++ptr;
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
ptr = goNum( ++ptr );
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Analyze the decimal part of a real number.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non numerical after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* fraqValue( char* ptr ) {
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
ptr = goNum( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a numerical value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type: JSON_REAL or JSON_INTEGER.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* numValue( char* ptr, json_t* property ) {
|
||||
if ( *ptr == '-' ) ++ptr;
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
if ( *ptr != '0' ) {
|
||||
ptr = goNum( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
else if ( isdigit( *++ptr ) ) return 0;
|
||||
property->type = JSON_INTEGER;
|
||||
if ( *ptr == '.' ) {
|
||||
ptr = fraqValue( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_REAL;
|
||||
}
|
||||
if ( *ptr == 'e' || *ptr == 'E' ) {
|
||||
ptr = expValue( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_REAL;
|
||||
}
|
||||
if ( !isEndOfPrimitive( *ptr ) ) return 0;
|
||||
if ( JSON_INTEGER == property->type ) {
|
||||
char const* value = property->u.value;
|
||||
bool const negative = *value == '-';
|
||||
static char const min[] = "-9223372036854775808";
|
||||
static char const max[] = "9223372036854775807";
|
||||
unsigned int const maxdigits = ( negative? sizeof min: sizeof max ) - 1;
|
||||
unsigned int const len = ptr - value;
|
||||
if ( len > maxdigits ) return 0;
|
||||
if ( len == maxdigits ) {
|
||||
char const tmp = *ptr;
|
||||
*ptr = '\0';
|
||||
char const* const threshold = negative ? min: max;
|
||||
if ( 0 > strcmp( threshold, value ) ) return 0;
|
||||
*ptr = tmp;
|
||||
}
|
||||
}
|
||||
ptr = setToNull( ptr );
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Add a property to a JSON object or array.
|
||||
* @param obj The handler of the JSON object or array.
|
||||
* @param property The handler of the property to be added. */
|
||||
static void add( json_t* obj, json_t* property ) {
|
||||
property->sibling = 0;
|
||||
if ( !obj->u.c.child ){
|
||||
obj->u.c.child = property;
|
||||
obj->u.c.last_child = property;
|
||||
} else {
|
||||
obj->u.c.last_child->sibling = property;
|
||||
obj->u.c.last_child = property;
|
||||
}
|
||||
}
|
||||
|
||||
/** Parser a string to get a json object value.
|
||||
* @param str Pointer to first character.
|
||||
* @param pool The handler of a json pool for creating json instances.
|
||||
* @retval Pointer to first character after the value. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* objValue( char* ptr, json_t* obj, jsonPool_t* pool ) {
|
||||
obj->type = JSON_OBJ;
|
||||
obj->u.c.child = 0;
|
||||
obj->sibling = 0;
|
||||
ptr++;
|
||||
for(;;) {
|
||||
ptr = goBlank( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
if ( *ptr == ',' ) {
|
||||
++ptr;
|
||||
continue;
|
||||
}
|
||||
char const endchar = ( obj->type == JSON_OBJ )? '}': ']';
|
||||
if ( *ptr == endchar ) {
|
||||
*ptr = '\0';
|
||||
json_t* parentObj = obj->sibling;
|
||||
if ( !parentObj ) return ++ptr;
|
||||
obj->sibling = 0;
|
||||
obj = parentObj;
|
||||
++ptr;
|
||||
continue;
|
||||
}
|
||||
json_t* property = pool->alloc( pool );
|
||||
if ( !property ) return 0;
|
||||
if( obj->type != JSON_ARRAY ) {
|
||||
if ( *ptr != '\"' ) return 0;
|
||||
ptr = propertyName( ptr, property );
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
else property->name = 0;
|
||||
add( obj, property );
|
||||
property->u.value = ptr;
|
||||
switch( *ptr ) {
|
||||
case '{':
|
||||
property->type = JSON_OBJ;
|
||||
property->u.c.child = 0;
|
||||
property->sibling = obj;
|
||||
obj = property;
|
||||
++ptr;
|
||||
break;
|
||||
case '[':
|
||||
property->type = JSON_ARRAY;
|
||||
property->u.c.child = 0;
|
||||
property->sibling = obj;
|
||||
obj = property;
|
||||
++ptr;
|
||||
break;
|
||||
case '\"': ptr = textValue( ptr, property ); break;
|
||||
case 't': ptr = trueValue( ptr, property ); break;
|
||||
case 'f': ptr = falseValue( ptr, property ); break;
|
||||
case 'n': ptr = nullValue( ptr, property ); break;
|
||||
default: ptr = numValue( ptr, property ); break;
|
||||
}
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Initialize a json pool.
|
||||
* @param pool The handler of the pool.
|
||||
* @return a instance of a json. */
|
||||
static json_t* poolInit( jsonPool_t* pool ) {
|
||||
jsonStaticPool_t *spool = json_containerOf( pool, jsonStaticPool_t, pool );
|
||||
spool->nextFree = 1;
|
||||
return spool->mem;
|
||||
}
|
||||
|
||||
/** Create an instance of a json from a pool.
|
||||
* @param pool The handler of the pool.
|
||||
* @retval The handler of the new instance if success.
|
||||
* @retval Null pointer if the pool was empty. */
|
||||
static json_t* poolAlloc( jsonPool_t* pool ) {
|
||||
jsonStaticPool_t *spool = json_containerOf( pool, jsonStaticPool_t, pool );
|
||||
if ( spool->nextFree >= spool->qty ) return 0;
|
||||
return spool->mem + spool->nextFree++;
|
||||
}
|
||||
|
||||
/** Checks whether an character belongs to set.
|
||||
* @param ch Character value to be checked.
|
||||
* @param set Set of characters. It is just a null-terminated string.
|
||||
* @return true or false there is membership or not. */
|
||||
static bool isOneOfThem( char ch, char const* set ) {
|
||||
while( *set != '\0' )
|
||||
if ( ch == *set++ )
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Increases a pointer while it points to a character that belongs to a set.
|
||||
* @param str The initial pointer value.
|
||||
* @param set Set of characters. It is just a null-terminated string.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goWhile( char* str, char const* set ) {
|
||||
for(; *str != '\0'; ++str ) {
|
||||
if ( !isOneOfThem( *str, set ) )
|
||||
return str;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Set of characters that defines a blank. */
|
||||
static char const* const blank = " \n\r\t\f";
|
||||
|
||||
/** Increases a pointer while it points to a white space character.
|
||||
* @param str The initial pointer value.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goBlank( char* str ) {
|
||||
return goWhile( str, blank );
|
||||
}
|
||||
|
||||
/** Increases a pointer while it points to a decimal digit character.
|
||||
* @param str The initial pointer value.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goNum( char* str ) {
|
||||
for( ; *str != '\0'; ++str ) {
|
||||
if ( !isdigit( *str ) )
|
||||
return str;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Set of characters that defines the end of an array or a JSON object. */
|
||||
static char const* const endofblock = "}]";
|
||||
|
||||
/** Set a char to '\0' and increase its pointer if the char is different to '}' or ']'.
|
||||
* @param ch Pointer to character.
|
||||
* @return Final value pointer. */
|
||||
static char* setToNull( char* ch ) {
|
||||
if ( !isOneOfThem( *ch, endofblock ) ) *ch++ = '\0';
|
||||
return ch;
|
||||
}
|
||||
|
||||
/** Indicate if a character is the end of a primitive value. */
|
||||
static bool isEndOfPrimitive( char ch ) {
|
||||
return ch == ',' || isOneOfThem( ch, blank ) || isOneOfThem( ch, endofblock );
|
||||
}
|
||||
|
||||
/** Add a character at the end of a string.
|
||||
* @param dest Pointer to the null character of the string
|
||||
* @param ch Value to be added.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* chtoa( char* dest, char ch ) {
|
||||
*dest = ch;
|
||||
*++dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Copy a null-terminated string.
|
||||
* @param dest Destination memory block.
|
||||
* @param src Source string.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* atoa( char* dest, char const* src ) {
|
||||
for( ; *src != '\0'; ++dest, ++src )
|
||||
*dest = *src;
|
||||
*dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Open a JSON object in a JSON string. */
|
||||
char* json_objOpen( char* dest, char const* name ) {
|
||||
if ( NULL == name )
|
||||
dest = chtoa( dest, '{' );
|
||||
else {
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":{" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Close a JSON object in a JSON string. */
|
||||
char* json_objClose( char* dest ) {
|
||||
if ( dest[-1] == ',' )
|
||||
--dest;
|
||||
return atoa( dest, "}," );
|
||||
}
|
||||
|
||||
/* Open an array in a JSON string. */
|
||||
char* json_arrOpen( char* dest, char const* name ) {
|
||||
if ( NULL == name )
|
||||
dest = chtoa( dest, '[' );
|
||||
else {
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":[" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Close an array in a JSON string. */
|
||||
char* json_arrClose( char* dest ) {
|
||||
if ( dest[-1] == ',' )
|
||||
--dest;
|
||||
return atoa( dest, "]," );
|
||||
}
|
||||
|
||||
/** Add the name of a text property.
|
||||
* @param dest Destination memory.
|
||||
* @param name The name of the property.
|
||||
* @return Pointer to the next char. */
|
||||
static char* strname( char* dest, char const* name ) {
|
||||
dest = chtoa( dest, '\"' );
|
||||
if ( NULL != name ) {
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":\"" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Get the hexadecimal digit of the least significant nibble of a integer. */
|
||||
static int nibbletoch( int nibble ) {
|
||||
return "0123456789ABCDEF"[ nibble % 16u ];
|
||||
}
|
||||
|
||||
/** Get the escape character of a non-printable.
|
||||
* @param ch Character source.
|
||||
* @return The escape character or null character if error. */
|
||||
static int escape( int ch ) {
|
||||
static struct { char code; char ch; } const pair[] = {
|
||||
{ '\"', '\"' }, { '\\', '\\' }, { '/', '/' }, { 'b', '\b' },
|
||||
{ 'f', '\f' }, { 'n', '\n' }, { 'r', '\r' }, { 't', '\t' },
|
||||
};
|
||||
for( int i = 0; i < sizeof pair / sizeof *pair; ++i )
|
||||
if ( ch == pair[i].ch )
|
||||
return pair[i].code;
|
||||
return '\0';
|
||||
}
|
||||
|
||||
/** Copy a null-terminated string inserting escape characters if needed.
|
||||
* @param dest Destination memory block.
|
||||
* @param src Source string.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* atoesc( char* dest, char const* src ) {
|
||||
for( ; *src != '\0'; ++dest, ++src ) {
|
||||
if ( *src >= ' ' && *src != '\"' && *src != '\\' && *src != '/' )
|
||||
*dest = *src;
|
||||
else {
|
||||
*dest++ = '\\';
|
||||
int const esc = escape( *src );
|
||||
if ( esc )
|
||||
*dest = esc;
|
||||
else {
|
||||
*dest++ = 'u';
|
||||
*dest++ = '0';
|
||||
*dest++ = '0';
|
||||
*dest++ = nibbletoch( *src / 16 );
|
||||
*dest++ = nibbletoch( *src );
|
||||
}
|
||||
}
|
||||
}
|
||||
*dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a text property in a JSON string. */
|
||||
char* json_str( char* dest, char const* name, char const* value ) {
|
||||
dest = strname( dest, name );
|
||||
dest = atoesc( dest, value );
|
||||
dest = atoa( dest, "\"," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Add the name of a primitive property.
|
||||
* @param dest Destination memory.
|
||||
* @param name The name of the property.
|
||||
* @return Pointer to the next char. */
|
||||
static char* primitivename( char* dest, char const* name ) {
|
||||
if( NULL == name )
|
||||
return dest;
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":" );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a boolean property in a JSON string. */
|
||||
char* json_bool( char* dest, char const* name, int value ) {
|
||||
dest = primitivename( dest, name );
|
||||
dest = atoa( dest, value ? "true," : "false," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a null property in a JSON string. */
|
||||
char* json_null( char* dest, char const* name ) {
|
||||
dest = primitivename( dest, name );
|
||||
dest = atoa( dest, "null," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Used to finish the root JSON object. After call json_objClose(). */
|
||||
char* json_end( char* dest ) {
|
||||
if ( ',' == dest[-1] ) {
|
||||
dest[-1] = '\0';
|
||||
--dest;
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
#define ALL_TYPES \
|
||||
X( json_int, int, "%d" ) \
|
||||
X( json_long, long, "%ld" ) \
|
||||
X( json_uint, unsigned int, "%u" ) \
|
||||
X( json_ulong, unsigned long, "%lu" ) \
|
||||
X( json_verylong, long long, "%lld" ) \
|
||||
X( json_double, double, "%g" ) \
|
||||
|
||||
|
||||
#define json_num( funcname, type, fmt ) \
|
||||
char* funcname( char* dest, char const* name, type value ) { \
|
||||
dest = primitivename( dest, name ); \
|
||||
dest += sprintf( dest, fmt, value ); \
|
||||
dest = chtoa( dest, ',' ); \
|
||||
return dest; \
|
||||
}
|
||||
|
||||
#define X( name, type, fmt ) json_num( name, type, fmt )
|
||||
ALL_TYPES
|
||||
#undef X
|
||||
Vendored
+270
@@ -0,0 +1,270 @@
|
||||
|
||||
/*
|
||||
|
||||
<https://github.com/rafagafe/tiny-json>
|
||||
|
||||
Licensed under the MIT License <http://opensource.org/licenses/MIT>.
|
||||
SPDX-License-Identifier: MIT
|
||||
Copyright (c) 2016-2018 Rafa Garcia <rafagarcia77@gmail.com>.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
*/
|
||||
|
||||
#ifndef _TINY_JSON_H_
|
||||
#define _TINY_JSON_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#define json_containerOf( ptr, type, member ) \
|
||||
((type*)( (char*)ptr - offsetof( type, member ) ))
|
||||
|
||||
/** @defgroup tinyJson Tiny JSON parser.
|
||||
* @{ */
|
||||
|
||||
/** Enumeration of codes of supported JSON properties types. */
|
||||
typedef enum {
|
||||
JSON_OBJ, JSON_ARRAY, JSON_TEXT, JSON_BOOLEAN,
|
||||
JSON_INTEGER, JSON_REAL, JSON_NULL
|
||||
} jsonType_t;
|
||||
|
||||
/** Structure to handle JSON properties. */
|
||||
typedef struct json_s {
|
||||
struct json_s* sibling;
|
||||
const char* name;
|
||||
union {
|
||||
const char* value;
|
||||
struct {
|
||||
struct json_s* child;
|
||||
struct json_s* last_child;
|
||||
} c;
|
||||
} u;
|
||||
jsonType_t type;
|
||||
} json_t;
|
||||
|
||||
/** Parse a string to get a json.
|
||||
* @param str String pointer with a JSON object. It will be modified.
|
||||
* @param mem Array of json properties to allocate.
|
||||
* @param qty Number of elements of mem.
|
||||
* @retval Null pointer if any was wrong in the parse process.
|
||||
* @retval If the parser process was successfully a valid handler of a json.
|
||||
* This property is always unnamed and its type is JSON_OBJ. */
|
||||
const json_t* json_create(char* str, json_t mem[], unsigned int qty);
|
||||
|
||||
/** Get the name of a json property.
|
||||
* @param json A valid handler of a json property.
|
||||
* @retval Pointer to null-terminated if property has name.
|
||||
* @retval Null pointer if the property is unnamed. */
|
||||
static inline const char* json_getName(const json_t* json) {
|
||||
return json->name;
|
||||
}
|
||||
|
||||
/** Get the value of a json property.
|
||||
* The type of property cannot be JSON_OBJ or JSON_ARRAY.
|
||||
* @param json A valid handler of a json property.
|
||||
* @return Pointer to null-terminated string with the value. */
|
||||
static inline const char* json_getValue(const json_t* property) {
|
||||
return property->u.value;
|
||||
}
|
||||
|
||||
/** Get the type of a json property.
|
||||
* @param json A valid handler of a json property.
|
||||
* @return The code of type.*/
|
||||
static inline jsonType_t json_getType(const json_t* json) {
|
||||
return json->type;
|
||||
}
|
||||
|
||||
/** Get the next sibling of a JSON property that is within a JSON object or array.
|
||||
* @param json A valid handler of a json property.
|
||||
* @retval The handler of the next sibling if found.
|
||||
* @retval Null pointer if the json property is the last one. */
|
||||
static inline const json_t* json_getSibling(const json_t* json) {
|
||||
return json->sibling;
|
||||
}
|
||||
|
||||
/** Search a property by its name in a JSON object.
|
||||
* @param obj A valid handler of a json object. Its type must be JSON_OBJ.
|
||||
* @param property The name of property to get.
|
||||
* @retval The handler of the json property if found.
|
||||
* @retval Null pointer if not found. */
|
||||
const json_t* json_getProperty(const json_t* obj, const char* property);
|
||||
|
||||
|
||||
/** Search a property by its name in a JSON object and return its value.
|
||||
* @param obj A valid handler of a json object. Its type must be JSON_OBJ.
|
||||
* @param property The name of property to get.
|
||||
* @retval If found a pointer to null-terminated string with the value.
|
||||
* @retval Null pointer if not found or it is an array or an object. */
|
||||
const char* json_getPropertyValue(const json_t* obj, const char* property);
|
||||
|
||||
/** Get the first property of a JSON object or array.
|
||||
* @param json A valid handler of a json property.
|
||||
* Its type must be JSON_OBJ or JSON_ARRAY.
|
||||
* @retval The handler of the first property if there is.
|
||||
* @retval Null pointer if the json object has not properties. */
|
||||
static inline const json_t* json_getChild(const json_t* json) {
|
||||
return json->u.c.child;
|
||||
}
|
||||
|
||||
/** Get the value of a json boolean property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_BOOLEAN.
|
||||
* @return The value stdbool. */
|
||||
static inline bool json_getBoolean(const json_t* property) {
|
||||
return *property->u.value == 't';
|
||||
}
|
||||
|
||||
/** Get the value of a json integer property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_INTEGER.
|
||||
* @return The value stdint. */
|
||||
static inline int64_t json_getInteger(const json_t* property) {
|
||||
return atoll( property->u.value );
|
||||
}
|
||||
|
||||
/** Get the value of a json real property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_REAL.
|
||||
* @return The value. */
|
||||
static inline double json_getReal(const json_t* property) {
|
||||
return atof( property->u.value );
|
||||
}
|
||||
|
||||
|
||||
/** Structure to handle a heap of JSON properties. */
|
||||
typedef struct jsonPool_s jsonPool_t;
|
||||
struct jsonPool_s {
|
||||
json_t* (*init)( jsonPool_t* pool );
|
||||
json_t* (*alloc)( jsonPool_t* pool );
|
||||
};
|
||||
|
||||
/** Parse a string to get a json.
|
||||
* @param str String pointer with a JSON object. It will be modified.
|
||||
* @param pool Custom json pool pointer.
|
||||
* @retval Null pointer if any was wrong in the parse process.
|
||||
* @retval If the parser process was successfully a valid handler of a json.
|
||||
* This property is always unnamed and its type is JSON_OBJ. */
|
||||
const json_t* json_createWithPool(char* str, jsonPool_t* pool);
|
||||
|
||||
/** @ } */
|
||||
|
||||
/** @defgroup makejoson Make JSON.
|
||||
* @{ */
|
||||
|
||||
/** Open a JSON object in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_objOpen(char* dest, const char* name);
|
||||
|
||||
/** Close a JSON object in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_objClose(char* dest);
|
||||
|
||||
/** Used to finish the root JSON object. After call json_objClose().
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_end(char* dest);
|
||||
|
||||
/** Open an array in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_arrOpen(char* dest, const char* name);
|
||||
|
||||
/** Close an array in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_arrClose(char* dest);
|
||||
|
||||
/** Add a text property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value A valid null-terminated string with the value.
|
||||
* Backslash escapes will be added for special characters.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_str(char* dest, const char* name, const char* value);
|
||||
|
||||
/** Add a boolean property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Zero for false. Non zero for true.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_bool(char* dest, const char* name, int value);
|
||||
|
||||
/** Add a null property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_null(char* dest, const char* name);
|
||||
|
||||
/** Add an integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_int(char* dest, const char* name, int value);
|
||||
|
||||
/** Add an unsigned integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_uint(char* dest, const char* name, unsigned int value);
|
||||
|
||||
/** Add a long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_long(char* dest, const char* name, long int value);
|
||||
|
||||
/** Add an unsigned long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_ulong(char* dest, const char* name, unsigned long int value);
|
||||
|
||||
/** Add a long long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_verylong(char* dest, const char* name, long long int value);
|
||||
|
||||
/** Add a double precision number property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_double(char* dest, const char* name, double value);
|
||||
|
||||
/** @ } */
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* _TINY_JSON_H_ */
|
||||
@@ -217,6 +217,14 @@ def print_man_environment_tail():
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored. These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
@@ -393,7 +401,7 @@ def print_parse_argloader_options(options):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "("
|
||||
conversion_func = "std::move("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
@@ -125,21 +125,36 @@ def parse_ops(ops):
|
||||
|
||||
RHS = EqualSplit[0].strip()
|
||||
if len(EqualSplit) > 1:
|
||||
OpDef.HasDest = True
|
||||
LHS = EqualSplit[0].strip()
|
||||
RHS = EqualSplit[1].strip()
|
||||
|
||||
# Parse the destination, must be one type of SSA, GPR, or FPR
|
||||
ResultType = EqualSplit[0].strip()
|
||||
if ResultType == "SSA":
|
||||
OpDef.DestType = "SSA" # We don't know this type right now
|
||||
elif ResultType == "GPR":
|
||||
OpDef.DestType = "GPR"
|
||||
elif ResultType == "GPRPair":
|
||||
OpDef.DestType = "GPRPair"
|
||||
elif ResultType == "FPR":
|
||||
OpDef.DestType = "FPR"
|
||||
if ":" in LHS:
|
||||
# Named destinations. This is a hack, but so is the entire
|
||||
# multi-destination support bolten onto the old IR...
|
||||
#
|
||||
# Named destinations require side effects because they break
|
||||
# SSA hard. Validate that.
|
||||
assert("HasSideEffects" in op_val and op_val["HasSideEffects"])
|
||||
|
||||
for Dest in LHS.split(","):
|
||||
Dest = Dest.strip()
|
||||
DType, Name = Dest.split(":$")
|
||||
|
||||
# If the destination appears also as a source, it is
|
||||
# read-modify-write.
|
||||
if Dest in RHS:
|
||||
# Turn RMW into an in/out source
|
||||
RHS = RHS.replace(Dest.strip(), f"{DType}:$Inout{Name}")
|
||||
else:
|
||||
# Turn named destinations into an out source.
|
||||
RHS += f", {DType}:$Out{Name}"
|
||||
else:
|
||||
ExitError("Unknown destination class type {}. Needs to be one of {SSA, GPR, GPRPair, FPR}".format(ResultType))
|
||||
# Single anonymous destination
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR")
|
||||
|
||||
OpDef.HasDest = True
|
||||
OpDef.DestType = LHS
|
||||
|
||||
# IR Op needs to start with a name
|
||||
RHS = RHS.split(" ", 1)
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
include(GNUInstallDirs)
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
@@ -85,7 +86,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
@@ -118,7 +118,6 @@ set (SRCS
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
Interface/Core/X86Tables/H0F38Tables.cpp
|
||||
Interface/Core/X86Tables/H0F3ATables.cpp
|
||||
Interface/Core/X86Tables/PrimaryGroupTables.cpp
|
||||
@@ -128,7 +127,6 @@ set (SRCS
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
@@ -139,7 +137,6 @@ set (SRCS
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
Utils/Telemetry.cpp
|
||||
@@ -181,7 +178,11 @@ endif()
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
|
||||
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
list (APPEND LIBS vixl)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
@@ -192,14 +193,6 @@ else()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
@@ -369,11 +362,30 @@ AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
install(TARGETS ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
|
||||
# Meta-library to link jemalloc libraries enabled in the build configuration.
|
||||
# Only needed for targets that run emulation. For others, use JemallocDummy.
|
||||
add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
|
||||
if (ENABLE_JEMALLOC)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_JEMALLOC=1 JEMALLOC_NO_RENAME=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC FEX_jemalloc)
|
||||
endif()
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
|
||||
target_link_libraries(JemallocLibs INTERFACE FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
# Dummy project to use for host tools.
|
||||
# This overrides use of jemalloc in FEXCore with the normal glibc allocator.
|
||||
add_library(JemallocDummy STATIC Utils/AllocatorHooks.cpp)
|
||||
target_include_directories(JemallocDummy PRIVATE "${PROJECT_SOURCE_DIR}/include/")
|
||||
endif()
|
||||
|
||||
# The shared library should always link enabled jemalloc libraries
|
||||
target_link_libraries(${PROJECT_NAME}_shared JemallocLibs)
|
||||
@@ -55,7 +55,7 @@ void JITSymbols::RegisterJITSpace(const void* HostAddr, uint32_t CodeSize) {
|
||||
}
|
||||
|
||||
// Buffered JIT symbols.
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
void JITSymbols::Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -79,7 +79,7 @@ void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, u
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
void JITSymbols::Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -104,7 +104,7 @@ void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, u
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
void JITSymbols::RegisterNamedRegion(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -128,7 +128,7 @@ void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void*
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
void JITSymbols::WriteBuffer(FEXCore::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
auto Now = std::chrono::steady_clock::now();
|
||||
if (!ForceWrite) {
|
||||
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) && Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
|
||||
|
||||
@@ -11,6 +11,26 @@
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
constexpr static std::chrono::milliseconds MAXIMUM_THRESHOLD {100};
|
||||
|
||||
JITSymbolBuffer()
|
||||
: LastWrite {std::chrono::steady_clock::now()} {}
|
||||
// stead_clock to ensure a monotonic increasing clock.
|
||||
// In highly stressed situations this can still cause >2% CPU time in vdso_clock_gettime.
|
||||
// If we need lower CPU time when JIT symbols are enabled then FEX can read the cycle counter directly.
|
||||
std::chrono::steady_clock::time_point LastWrite {};
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
JITSymbols();
|
||||
@@ -21,16 +41,16 @@ public:
|
||||
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
|
||||
|
||||
// Allocate JIT buffer.
|
||||
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<Core::JITSymbolBuffer>();
|
||||
static fextl::unique_ptr<FEXCore::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<FEXCore::JITSymbolBuffer>();
|
||||
}
|
||||
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
|
||||
private:
|
||||
int fd {-1};
|
||||
void WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
void WriteBuffer(FEXCore::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -160,7 +160,31 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(state, lhs, rhs);
|
||||
/*
|
||||
* FPREM is not an IEEE-754 remainder. From the spec:
|
||||
*
|
||||
* Computes the remainder obtained from dividing the value in the ST(0)
|
||||
* register (the dividend) by the value in the ST(1) register (the divisor
|
||||
* or modulus), and stores the result in ST(0). The remainder represents the
|
||||
* following value:
|
||||
*
|
||||
* Remainder := ST(0) − (Q * ST(1))
|
||||
*
|
||||
* Here, Q is an integer value that is obtained by truncating the
|
||||
* floating-point number quotient of [ST(0) / ST(1)] toward zero.
|
||||
*
|
||||
* We implement this sequence literally. softfloat_round_minMag means
|
||||
* "truncate towards zero".
|
||||
*/
|
||||
extFloat80_t quotient = extF80_div(state, lhs, rhs);
|
||||
extFloat80_t Q = extF80_roundToInt(state, quotient, softfloat_round_minMag, true);
|
||||
bool Q_zero = Q.signif == 0 && (Q.signExp & ~(1 << 15)) == 0;
|
||||
|
||||
if (Q_zero) {
|
||||
return lhs;
|
||||
} else {
|
||||
return extF80_sub(state, lhs, extF80_mul(state, Q, rhs));
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -262,6 +286,10 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
extFloat80_t Zero {0, 0};
|
||||
if (extF80_eq(state, lhs, Zero)) {
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Int = FRNDINT(state, rhs, softfloat_round_minMag);
|
||||
LIBRARY_PRECISION Src2_d = Int.ToFMax(state);
|
||||
Src2_d = exp2l(Src2_d);
|
||||
|
||||
@@ -43,7 +43,8 @@ namespace DefaultValues {
|
||||
} // namespace DefaultValues
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
PATH_DATA_DIR_GLOBAL,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
@@ -53,8 +54,8 @@ enum Paths {
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
void SetDataDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_DATA_DIR_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
@@ -73,15 +74,15 @@ const fextl::string& GetTelemetryDirectory() {
|
||||
Path = TelemetryDirectory;
|
||||
Path += "/";
|
||||
} else {
|
||||
Path = Config::GetDataDirectory() + "Telemetry/";
|
||||
Path = Config::GetDataDirectory(false) + "Telemetry/";
|
||||
}
|
||||
}
|
||||
|
||||
return Path;
|
||||
}
|
||||
|
||||
const fextl::string& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
const fextl::string& GetDataDirectory(bool Global) {
|
||||
return Paths[PATH_DATA_DIR_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetConfigDirectory(bool Global) {
|
||||
@@ -230,27 +231,26 @@ void Load() {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, fextl::string PathName) {
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
|
||||
// Expand home if it exists
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
PathName.replace(0, 1, Home);
|
||||
return PathName;
|
||||
if (PathName.starts_with("~/")) {
|
||||
Home.append(PathName.begin() + 1, PathName.end());
|
||||
return Home;
|
||||
}
|
||||
|
||||
// Expand relative path to absolute
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char* RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
if (RealPath && FHU::Filesystem::Exists(RealPath)) {
|
||||
return RealPath;
|
||||
}
|
||||
|
||||
// Only return if it exists
|
||||
@@ -318,81 +318,61 @@ fextl::string FindContainerPrefix() {
|
||||
void ReloadMetaLayer() {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 0;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
}
|
||||
|
||||
fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
const fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, const fextl::string& PathName) {
|
||||
const auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_ROOTFS);
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
fextl::string NamedRootFS = GetDataDirectory(false) + "RootFS/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, *PathName);
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, *PathName);
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKCONFIG);
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
fextl::string NamedConfig = GetDataDirectory(false) + "ThunkConfigs/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
|
||||
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
|
||||
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_OUTPUTLOG);
|
||||
if (*PathName != "stdout" && *PathName != "stderr" && *PathName != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, *PathName);
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) && !FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
FEX_CONFIG_OPT(PathName, DUMPIR);
|
||||
if (PathName() != "no") {
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_DUMPIR);
|
||||
if (*PathName != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
|
||||
@@ -1,19 +1,6 @@
|
||||
{
|
||||
"Options": {
|
||||
"CPU": {
|
||||
"Core": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
|
||||
"TextDefault": "irjit",
|
||||
"ShortArg": "c",
|
||||
"Choices": [ "irjit", "host" ],
|
||||
"ArgumentHandler": "CoreHandler",
|
||||
"Desc": [
|
||||
"Which CPU core to use",
|
||||
"host only exists on x86_64",
|
||||
"[irjit, host]"
|
||||
]
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -142,7 +129,7 @@
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks/",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
@@ -158,7 +145,7 @@
|
||||
},
|
||||
"ThunkHostLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks_32/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit host-side thunking libraries."
|
||||
]
|
||||
@@ -422,6 +409,14 @@
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
]
|
||||
},
|
||||
"StrictInProcessSplitLocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -509,6 +504,13 @@
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
},
|
||||
"NeedsSeccomp": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
@@ -39,10 +40,6 @@ void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThrea
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator* _SignalDelegation) {
|
||||
SignalDelegation = _SignalDelegation;
|
||||
}
|
||||
@@ -52,6 +49,10 @@ void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandl
|
||||
SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetThunkHandler(FEXCore::ThunkHandler* Handler) {
|
||||
ThunkHandler = Handler;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
@@ -26,14 +26,8 @@
|
||||
#include <stdint.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <queue>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
@@ -65,18 +59,21 @@ namespace Validation {
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
enum CoreRunningMode {
|
||||
MODE_RUN = 0,
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct ExitFunctionLinkData {
|
||||
uint64_t HostBranch;
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
struct CustomIRResult {
|
||||
void* Creator;
|
||||
void* Data;
|
||||
|
||||
CustomIRResult(void* Creator, void* Data)
|
||||
: Creator(Creator)
|
||||
, Data(Data) {}
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE = 128;
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
@@ -94,8 +91,6 @@ public:
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
@@ -134,7 +129,7 @@ public:
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
@@ -152,6 +147,7 @@ public:
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator* SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) override;
|
||||
void SetThunkHandler(FEXCore::ThunkHandler* Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
@@ -191,9 +187,10 @@ public:
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
std::optional<CustomIRResult>
|
||||
AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
|
||||
void AppendThunkDefinitions(std::span<const FEXCore::IR::ThunkDefinition> Definitions) override;
|
||||
void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
@@ -204,8 +201,8 @@ public:
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck {false};
|
||||
@@ -216,17 +213,15 @@ public:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
@@ -237,28 +232,25 @@ public:
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
} Config;
|
||||
|
||||
|
||||
std::atomic_bool CoreShuttingDown {false};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
// CPUID depends on HostFeatures so needs to be initialized after that.
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
@@ -326,16 +318,6 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
@@ -344,27 +326,21 @@ public:
|
||||
return AtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
// If atomic-based TSO emulation is enabled for vector operations.
|
||||
bool IsVectorAtomicTSOEnabled() const {
|
||||
return VectorAtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
// If atomic-based TSO emulation is enabled for memcpy operations.
|
||||
bool IsMemcpyAtomicTSOEnabled() const {
|
||||
return MemcpyAtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
// Returns if Software TSO emulation is required.
|
||||
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
|
||||
// This will still return true if on a single thread and TSO is currently disabled.
|
||||
//
|
||||
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
|
||||
// we return consistent results.
|
||||
//
|
||||
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
|
||||
bool SoftwareTSORequired() const {
|
||||
if (SupportsHardwareTSO) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return Config.TSOEnabled;
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override {
|
||||
ExitOnHLT = true;
|
||||
}
|
||||
@@ -378,9 +354,15 @@ protected:
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
|
||||
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
|
||||
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -403,6 +385,9 @@ private:
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool VectorAtomicTSOEmulationEnabled = false;
|
||||
bool MemcpyAtomicTSOEmulationEnabled = false;
|
||||
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -29,11 +28,12 @@ namespace FEXCore::CPU {
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
// Note that rax/rdx are rearranged here so we can coalesce cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
@@ -194,12 +194,12 @@ namespace x64 {
|
||||
} // namespace x64
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
// All but x19 and x29 are caller saved. eax/edx rearranged for cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 10> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
@@ -373,6 +373,20 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::X86State::X86Reg Arm64Emitter::GetX86RegRelationToARMReg(ARMEmitter::Register Reg) {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
const auto& RegI = StaticRegisters[i];
|
||||
if (RegI == Reg) {
|
||||
// X86 Registers are mapped linerally from the StaticRegisters span.
|
||||
// Directly correlating Enum index to span index.
|
||||
return static_cast<FEXCore::X86State::X86Reg>(FEXCore::ToUnderlying(FEXCore::X86State::X86Reg::REG_RAX) + i);
|
||||
}
|
||||
}
|
||||
|
||||
// Unmapped register.
|
||||
return FEXCore::X86State::X86Reg::REG_INVALID;
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
@@ -4,11 +4,6 @@
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#endif
|
||||
@@ -17,6 +12,7 @@
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
@@ -106,6 +102,10 @@ protected:
|
||||
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
// Returning REG_INVALID if there was no mapping.
|
||||
FEXCore::X86State::X86Reg GetX86RegRelationToARMReg(ARMEmitter::Register Reg);
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
|
||||
@@ -1,51 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore {
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
std::fstream Output;
|
||||
Output.open("output.csv", std::fstream::out | std::fstream::binary);
|
||||
|
||||
if (!Output.is_open()) {
|
||||
return;
|
||||
}
|
||||
|
||||
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
|
||||
|
||||
for (auto it : SamplingMap) {
|
||||
if (!it.second->TotalCalls) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Output << "0x" << std::hex << it.first << ", " << std::dec << it.second->Min << ", " << std::dec << it.second->Max << ", " << std::dec
|
||||
<< it.second->TotalTime << ", " << std::dec << it.second->TotalCalls << ", " << std::dec
|
||||
<< ((double)it.second->TotalTime / (double)it.second->TotalCalls) << std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData* BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
auto it = SamplingMap.find(RIP);
|
||||
if (it != SamplingMap.end()) {
|
||||
return it->second;
|
||||
}
|
||||
BlockData* NewData = new BlockData {};
|
||||
memset(NewData, 0, sizeof(BlockData));
|
||||
NewData->Min = ~0ULL;
|
||||
SamplingMap[RIP] = NewData;
|
||||
return NewData;
|
||||
}
|
||||
|
||||
BlockSamplingData::~BlockSamplingData() {
|
||||
DumpBlockData();
|
||||
for (auto it : SamplingMap) {
|
||||
delete it.second;
|
||||
}
|
||||
SamplingMap.clear();
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -1,25 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockSamplingData {
|
||||
public:
|
||||
struct BlockData {
|
||||
uint64_t Start, End;
|
||||
uint64_t Min, Max;
|
||||
uint64_t TotalTime;
|
||||
uint64_t TotalCalls;
|
||||
};
|
||||
|
||||
BlockData* GetBlockData(uint64_t RIP);
|
||||
~BlockSamplingData();
|
||||
|
||||
void DumpBlockData();
|
||||
|
||||
private:
|
||||
std::unordered_map<uint64_t, BlockData*> SamplingMap;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -48,11 +48,6 @@ namespace CPU {
|
||||
CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
/**
|
||||
* @return The name of this backend
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual fextl::string GetName() = 0;
|
||||
|
||||
struct CompiledCode {
|
||||
// Where this code block begins.
|
||||
@@ -124,9 +119,6 @@ namespace CPU {
|
||||
*
|
||||
* This is a thread specific compilation unit since there is one CPUBackend per guest thread
|
||||
*
|
||||
* If NeedsOpDispatch is returning false then IR and DebugData may be null and the expectation is that the code will still compile
|
||||
* FEXCore::Core::ThreadState* is valid at the time of compilation.
|
||||
*
|
||||
* @param IR - IR that maps to the IR for this RIP
|
||||
* @param DebugData - Debug data that is available for this IR indirectly
|
||||
*
|
||||
@@ -149,26 +141,6 @@ namespace CPU {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
*
|
||||
* @return Currently unused
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual void* MapRegion(void* HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
|
||||
*
|
||||
* This is useful if the FEXCore Frontend hits an x86-64 instruction that isn't understood but can continue regardless
|
||||
*
|
||||
* This is useful for example, a VM based CPUbackend
|
||||
*
|
||||
* @return true if it needs the IR
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual bool NeedsOpDispatch() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
/**
|
||||
|
||||
@@ -72,8 +72,18 @@ namespace ProductNames {
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
|
||||
static const char ARM_Firestorm[] = "Apple Firestorm";
|
||||
static const char ARM_Icestorm[] = "Apple Icestorm";
|
||||
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
|
||||
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
|
||||
static const char ARM_Firestorm_M1Pro[] = "Apple Firestorm (M1 Pro)";
|
||||
static const char ARM_Icestorm_M1Pro[] = "Apple Icestorm (M1 Pro)";
|
||||
static const char ARM_Firestorm_M1Max[] = "Apple Firestorm (M1 Max)";
|
||||
static const char ARM_Icestorm_M1Max[] = "Apple Icestorm (M1 Max)";
|
||||
static const char ARM_Avalanche_M2[] = "Apple Avalanche (M2)";
|
||||
static const char ARM_Blizzard_M2[] = "Apple Blizzard (M2)";
|
||||
static const char ARM_Avalanche_M2Pro[] = "Apple Avalanche (M2 Pro)";
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
#else
|
||||
@@ -86,20 +96,39 @@ static uint32_t GetCPUID() {
|
||||
return CPU;
|
||||
}
|
||||
|
||||
struct CPUFamily {
|
||||
uint32_t Stepping : 4;
|
||||
uint32_t Model : 4;
|
||||
uint32_t ExtendedModel : 4;
|
||||
uint32_t FamilyID : 4;
|
||||
uint32_t ExtendedFamilyID : 8;
|
||||
uint32_t ProcessorType : 4;
|
||||
};
|
||||
|
||||
constexpr static uint32_t GenerateFamily(const CPUFamily Family) {
|
||||
return Family.Stepping | (Family.Model << 4) | (Family.FamilyID << 8) | (Family.ProcessorType << 12) | (Family.ExtendedModel << 16) |
|
||||
(Family.ExtendedFamilyID << 20);
|
||||
}
|
||||
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = 0 | // Stepping
|
||||
(0xA << 4) | // Model
|
||||
(0xF << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(0 << 16) | // Extended model ID
|
||||
(1 << 20); // Extended family ID
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
.Stepping = 0,
|
||||
.Model = 0xA,
|
||||
.ExtendedModel = 0,
|
||||
.FamilyID = 0xF,
|
||||
.ExtendedFamilyID = 1,
|
||||
.ProcessorType = 0,
|
||||
});
|
||||
|
||||
#else
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = 0 | // Stepping
|
||||
(0x7 << 4) | // Model
|
||||
(0x6 << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(1 << 16) | // Extended model ID
|
||||
(0x0 << 20); // Extended family ID
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
.Stepping = 1,
|
||||
.Model = 6,
|
||||
.ExtendedModel = 0xA,
|
||||
.FamilyID = 6,
|
||||
.ExtendedFamilyID = 0,
|
||||
.ProcessorType = 0,
|
||||
});
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
@@ -114,27 +143,16 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
uint64_t MIDR {};
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec {};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
std::array<char, 18> Data;
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFileToBuffer(MIDRPath, Data) == sizeof(Data)) {
|
||||
uint64_t NewMIDR {};
|
||||
std::string_view MIDRView(Data.data(), sizeof(Data));
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
auto NewMIDR = CTX->HostFeatures.CPUMIDRs[i];
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
|
||||
struct CPUMIDR {
|
||||
@@ -147,11 +165,16 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 48> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
{0x61, 0x039, 1, ProductNames::ARM_Avalanche_M2Max}, // Apple Avalanche (M2 Max)
|
||||
{0x61, 0x035, 1, ProductNames::ARM_Avalanche_M2Pro}, // Apple Avalanche (M2 Pro)
|
||||
{0x61, 0x033, 1, ProductNames::ARM_Avalanche_M2}, // Apple Avalanche (M2)
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
@@ -193,7 +216,13 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd07, 1, ProductNames::ARM_A57}, // A57
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x61, 0x038, 0, ProductNames::ARM_Blizzard_M2Max}, // Apple Blizzard (M2 Max)
|
||||
{0x61, 0x034, 0, ProductNames::ARM_Blizzard_M2Pro}, // Apple Blizzard (M2 Pro)
|
||||
{0x61, 0x032, 0, ProductNames::ARM_Blizzard_M2}, // Apple Blizzard (M2)
|
||||
{0x61, 0x028, 0, ProductNames::ARM_Icestorm_M1Max}, // Apple Icestorm (M1 Max)
|
||||
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
|
||||
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
@@ -608,8 +637,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
// Disable Enhanced REP MOVS when TSO is enabled.
|
||||
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
|
||||
// This is due to LRCPC performance on Cortex being abysmal.
|
||||
// Only enable EnhancedREPMOVS if SoftwareTSO isn't required OR if MemcpySetTSO is not enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->SoftwareTSORequired() == false || MemcpySetTSOEnabled() == false;
|
||||
// Only enable EnhancedREPMOVS if atomic memcpy tso emulation isn't enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->IsMemcpyAtomicTSOEnabled() == false;
|
||||
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
|
||||
|
||||
// Number of subfunctions
|
||||
@@ -772,7 +801,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
if (FrequencyHz) {
|
||||
Res.eax = 1;
|
||||
Res.ebx = CTX->Config.SmallTSCScale() ? FEXCore::Context::TSC_SCALE : 1;
|
||||
Res.ebx = 1U << CTX->Config.TSCScale;
|
||||
Res.ecx = FrequencyHz;
|
||||
}
|
||||
return Res;
|
||||
@@ -1185,7 +1214,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {
|
||||
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
@@ -118,8 +118,6 @@ private:
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
@@ -20,7 +19,6 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -29,11 +27,13 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
@@ -78,9 +78,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
: HostFeatures {Features}
|
||||
, CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
@@ -94,8 +91,13 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
if (FEXCore::GetCycleCounterFrequency() >= FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
Config.SmallTSCScale = false;
|
||||
uint64_t FrequencyCounter = FEXCore::GetCycleCounterFrequency();
|
||||
if (FrequencyCounter && FrequencyCounter < FEXCore::Context::TSC_SCALE_MAXIMUM && Config.SmallTSCScale()) {
|
||||
// Scale TSC until it is at the minimum required.
|
||||
while (FrequencyCounter < FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
FrequencyCounter <<= 1;
|
||||
++Config.TSCScale;
|
||||
}
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
@@ -190,6 +192,9 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC)) & 1;
|
||||
|
||||
// CF is inverted in our representation, undo the invert here.
|
||||
CF ^= 1;
|
||||
|
||||
// Pack in to EFLAGS
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_RAW_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_RAW_LOC;
|
||||
@@ -293,10 +298,10 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate packed NZCV
|
||||
// Calculate packed NZCV. Note CF is inverted.
|
||||
uint32_t Packed_NZCV {};
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 0 : 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC);
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC) : 0;
|
||||
memcpy(&Frame->State.flags[X86State::RFLAG_NZCV_LOC], &Packed_NZCV, sizeof(Packed_NZCV));
|
||||
@@ -313,8 +318,6 @@ bool ContextImpl::InitCore() {
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.SupportsAVX = HostFeatures.SupportsAVX,
|
||||
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
|
||||
@@ -343,9 +346,8 @@ bool ContextImpl::InitCore() {
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
// WIN32 always needs the interrupt fault check to be enabled.
|
||||
#elif !defined(_M_ARM64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
@@ -383,9 +385,6 @@ void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
|
||||
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
@@ -411,20 +410,14 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM: Thread->CPUBackend = CustomCPUFactory(this, Thread); break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown core configuration"); break;
|
||||
}
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
@@ -468,8 +461,14 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
StrictSplitLockMutex = 0;
|
||||
}
|
||||
} else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::unlock(&StrictSplitLockMutex);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
@@ -477,6 +476,9 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::lock(&StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -595,6 +597,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
const FEXCore::X86Tables::DecodedInst* DecodedInfo {nullptr};
|
||||
@@ -971,7 +978,8 @@ void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
std::optional<CustomIRResult>
|
||||
ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
@@ -981,10 +989,50 @@ CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIR
|
||||
|
||||
if (!InsertedIterator.second) {
|
||||
const auto& [fn, Creator, Data] = InsertedIterator.first->second;
|
||||
return CustomIRResult(std::move(lk), Creator, Data);
|
||||
} else {
|
||||
lk.unlock();
|
||||
return CustomIRResult(std::move(lk), 0, 0);
|
||||
return CustomIRResult(Creator, Data);
|
||||
}
|
||||
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) {
|
||||
LOGMAN_THROW_AA_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
if (!Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}", Entrypoint, GuestThunkEntrypoint);
|
||||
|
||||
auto Result = AddCustomIREntrypoint(
|
||||
Entrypoint,
|
||||
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
if (GPRSize == 8) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), X86State::REG_R11, IR::GPRClass, GPRSize);
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
ThunkHandler, (void*)GuestThunkEntrypoint);
|
||||
|
||||
if (Result.has_value()) {
|
||||
if (Result->Creator != ThunkHandler) {
|
||||
ERROR_AND_DIE_FMT("Input address for AddThunkTrampoline is already linked by another module");
|
||||
}
|
||||
if (Result->Data != (void*)GuestThunkEntrypoint) {
|
||||
// NOTE: This may happen in Vulkan thunks if the Vulkan driver resolves two different symbols
|
||||
// to the same function (e.g. vkGetPhysicalDeviceFeatures2/vkGetPhysicalDeviceFeatures2KHR)
|
||||
LogMan::Msg::EFmt("Input address for AddThunkTrampoline is already linked elsewhere");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1007,12 +1055,6 @@ void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(std::span<const FEXCore::IR::ThunkDefinition> Definitions) {
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
|
||||
@@ -274,12 +274,10 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -557,12 +555,10 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -694,12 +690,8 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
}
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
|
||||
/* uint8_t P1 = */ ReadByte();
|
||||
/* uint8_t P2 = */ ReadByte();
|
||||
/* uint8_t P3 = */ ReadByte();
|
||||
uint8_t EVEXOp = ReadByte();
|
||||
return NormalOp(&EVEXTableOps[EVEXOp], EVEXOp);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
|
||||
@@ -967,8 +959,14 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
// If the target RIP is within the symbol ranges then we are golden
|
||||
if (TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress) {
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
#endif
|
||||
|
||||
if (ValidMultiblockMember) {
|
||||
// Update our conditional branch ranges before we return
|
||||
if (Conditional) {
|
||||
MaxCondBranchForward = std::max(MaxCondBranchForward, TargetRIP);
|
||||
@@ -977,12 +975,12 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
BlocksToDecode.insert(FallthroughRIP);
|
||||
CurrentBlockTargets.insert(FallthroughRIP);
|
||||
}
|
||||
}
|
||||
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
BlocksToDecode.insert(TargetRIP);
|
||||
CurrentBlockTargets.insert(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
@@ -1079,6 +1077,8 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1115,7 +1115,6 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) [[unlikely]] {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
@@ -1123,6 +1122,14 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
DecodeInst->InstSize = 0;
|
||||
}
|
||||
|
||||
if (!ErrorDuringDecoding) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
auto TableInfo = DecodedBuffer[BlockStartOffset + BlockNumberOfInstructions].TableInfo;
|
||||
if (!TableInfo || !TableInfo->OpcodeDispatcher) {
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, RIPToDecode + PCOffset + DecodeInst->InstSize);
|
||||
++TotalInstructions;
|
||||
@@ -1130,7 +1137,17 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
++DecodedSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) {
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) [[unlikely]] {
|
||||
if (!EntryBlock) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
TotalInstructions -= CurrentBlockDecoding.NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
BlockNumberOfInstructions = 0;
|
||||
InstStream -= PCOffset;
|
||||
CurrentBlockTargets.clear();
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1160,6 +1177,9 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
InstStream += DecodeInst->InstSize;
|
||||
}
|
||||
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
CurrentBlockTargets.clear();
|
||||
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
HasBlocks.emplace(RIPToDecode);
|
||||
|
||||
@@ -1167,6 +1187,8 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockInfo.TotalInstructionCount += BlockNumberOfInstructions;
|
||||
|
||||
EntryBlock = false;
|
||||
}
|
||||
|
||||
for (auto CodePage : CodePages) {
|
||||
|
||||
@@ -104,6 +104,7 @@ private:
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
|
||||
DecodedBlockInformation BlockInfo;
|
||||
fextl::set<uint64_t> CurrentBlockTargets;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW) {
|
||||
softfloat_state State;
|
||||
softfloat_state State {};
|
||||
State.detectTininess = softfloat_tininess_afterRounding;
|
||||
State.exceptionFlags = 0;
|
||||
|
||||
@@ -322,8 +322,11 @@ struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
double trunc = (double)(int64_t)(src2); // truncate
|
||||
return src1 * exp2(trunc);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
}
|
||||
double trun = trunc(src2);
|
||||
return src1 * exp2(trun);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -41,21 +42,6 @@ DEF_BINOP_WITH_CONSTANT(Lshl, lslv, lsl)
|
||||
DEF_BINOP_WITH_CONSTANT(Lshr, lsrv, lsr)
|
||||
DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
|
||||
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: {
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Src = GetRegPair(Op->Pair.ID());
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, Src.first);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.second, Src.second);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Constant) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
auto Dst = GetReg(Node);
|
||||
@@ -130,6 +116,21 @@ DEF_OP(AdcWithFlags) {
|
||||
adcs(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AdcZeroWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZeroWithFlags>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cset(Size, TMP1, ARMEmitter::Condition::CC_CC);
|
||||
adds(Size, GetReg(Node), GetReg(Op->Src1.ID()), TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(AdcZero) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZero>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cinc(Size, GetReg(Node), GetReg(Op->Src1.ID()), ARMEmitter::Condition::CC_CC);
|
||||
}
|
||||
|
||||
DEF_OP(Adc) {
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
|
||||
@@ -190,6 +191,30 @@ DEF_OP(TestNZ) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TestZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestZ>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < 4, "TestNZ used at higher sizes");
|
||||
const auto EmitSize = ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
uint64_t Mask = IROp->Size == 8 ? ~0ULL : ((1ull << (IROp->Size * 8)) - 1);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
|
||||
LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
if (Src1 == Src2) {
|
||||
tst(EmitSize, Src1 /* Src2 */, Mask);
|
||||
} else {
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
tst(EmitSize, TMP1, Mask);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
@@ -232,10 +257,8 @@ DEF_OP(CmpPairZ) {
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Compare, setting Z and clobbering NzCV
|
||||
const auto Src1 = GetRegPair(Op->Src1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Src2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, GetReg(Op->Src1Lo.ID()), GetReg(Op->Src2Lo.ID()));
|
||||
ccmp(EmitSize, GetReg(Op->Src1Hi.ID()), GetReg(Op->Src2Hi.ID()), ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
// Restore NzCV
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
@@ -274,8 +297,43 @@ DEF_OP(SetSmallNZV) {
|
||||
}
|
||||
|
||||
DEF_OP(AXFlag) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
|
||||
axflag();
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
axflag();
|
||||
} else {
|
||||
// AXFLAG is defined in the Arm spec as
|
||||
//
|
||||
// gt: nzCv -> nzCv
|
||||
// lt: Nzcv -> nzcv <==> 1 + 0
|
||||
// eq: nZCv -> nZCv <==> 1 + (~0)
|
||||
// un: nzCV -> nZcv <==> 0 + 0
|
||||
//
|
||||
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
|
||||
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
|
||||
auto V_inv = GetReg(IROp->Args[0].ID());
|
||||
csetm(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Condition::CC_EQ);
|
||||
ccmn(ARMEmitter::Size::i64Bit, V_inv, TMP1, ARMEmitter::StatusFlags {0x2} /* nzCv */, ARMEmitter::Condition::CC_LE);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Parity) {
|
||||
auto Op = IROp->C<IR::IROp_Parity>();
|
||||
auto Raw = GetReg(Op->Raw.ID());
|
||||
auto Dest = GetReg(Node);
|
||||
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, Raw, Raw, ARMEmitter::ShiftType::LSR, 4);
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 2);
|
||||
|
||||
if (Op->Invert) {
|
||||
eon(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
|
||||
} else {
|
||||
eor(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
|
||||
}
|
||||
|
||||
// The above sequence leaves garbage in the upper bits.
|
||||
if (Op->Mask) {
|
||||
and_(ARMEmitter::Size::i32Bit, Dest, Dest, 1);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
@@ -663,6 +721,11 @@ DEF_OP(ShiftFlags) {
|
||||
lsrv(EmitSize, CFWord, Src1, CFWord);
|
||||
}
|
||||
|
||||
if (Op->InvertCF) {
|
||||
mvn(ARMEmitter::Size::i64Bit, TMP1, CFWord);
|
||||
CFWord = TMP1;
|
||||
}
|
||||
|
||||
bool SetOF = Op->Shift != IR::ShiftType::ASR;
|
||||
if (SetOF) {
|
||||
// Only defined when Shift is 1 else undefined
|
||||
@@ -702,6 +765,50 @@ DEF_OP(ShiftFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RotateFlags) {
|
||||
auto Op = IROp->C<IR::IROp_RotateFlags>();
|
||||
const auto Result = GetReg(Op->Result.ID());
|
||||
const auto Shift = GetReg(Op->Shift.ID());
|
||||
const bool Left = Op->Left;
|
||||
const auto EmitSize = Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = Op->Size * 8;
|
||||
unsigned CFBit = Left ? 0 : BitSize - 1;
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
// For ROL, OF is the LSB and MSB XOR'd together.
|
||||
// OF is architecturally only defined for 1-bit rotate.
|
||||
eor(ARMEmitter::Size::i64Bit, TMP1, Result, Result, ARMEmitter::ShiftType::LSR, Left ? BitSize - 1 : 1);
|
||||
unsigned OFBit = Left ? 0 : BitSize - 2;
|
||||
|
||||
// Invert result so we get inverted carry.
|
||||
mvn(ARMEmitter::Size::i64Bit, TMP2, Result);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
rmif(TMP2, (CFBit - 1) % 64, 1 << 1 /* nzCv */);
|
||||
rmif(TMP1, OFBit, 1 << 0 /* nzcV */);
|
||||
} else {
|
||||
if (OFBit != 0) {
|
||||
lsr(EmitSize, TMP1, TMP1, OFBit);
|
||||
}
|
||||
if (CFBit != 0) {
|
||||
lsr(EmitSize, TMP2, TMP2, CFBit);
|
||||
}
|
||||
|
||||
mrs(TMP3, ARMEmitter::SystemRegister::NZCV);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP3, TMP1, 28 /* V */, 1);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP3, TMP2, 29 /* C */, 1);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1368,11 +1475,6 @@ DEF_OP(Select) {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetRegPair(Op->Cmp1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Cmp2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, cc);
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
@@ -1433,6 +1535,12 @@ DEF_OP(NZCVSelect) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectIncrement) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectIncrement>();
|
||||
|
||||
csinc(ConvertSize(IROp), GetReg(Node), GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), MapCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1443,6 +1551,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1463,7 +1572,6 @@ DEF_OP(VExtractToGPR) {
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host doesn't support SVE. Cannot perform 256-bit operation.");
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
|
||||
|
||||
@@ -7,7 +7,8 @@ $end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
@@ -15,19 +15,43 @@ DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Expected = GetRegPair(Op->Expected.ID());
|
||||
auto Desired = GetRegPair(Op->Desired.ID());
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
auto Expected0 = GetReg(Op->ExpectedLo.ID());
|
||||
auto Expected1 = GetReg(Op->ExpectedHi.ID());
|
||||
auto Desired0 = GetReg(Op->DesiredLo.ID());
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP3, Expected.first);
|
||||
mov(EmitSize, TMP4, Expected.second);
|
||||
// RA has heuristics to try to pair sources, but we need to handle the cases
|
||||
// where they fail. We do so by moving to temporaries. Note we use 64-bit
|
||||
// moves here even for 32-bit cmpxchg, for the Firestorm register renamer.
|
||||
if (Desired1.Idx() != (Desired0.Idx() + 1) || Desired0.Idx() & 1) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, Desired0);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, Desired1);
|
||||
Desired0 = TMP1;
|
||||
Desired1 = TMP2;
|
||||
}
|
||||
|
||||
caspal(EmitSize, TMP3, TMP4, Desired.first, Desired.second, MemSrc);
|
||||
mov(EmitSize, Dst.first, TMP3.R());
|
||||
mov(EmitSize, Dst.second, TMP4.R());
|
||||
auto CaspalDst0 = Dst0;
|
||||
auto CaspalDst1 = Dst1;
|
||||
if (CaspalDst1.Idx() != (CaspalDst0.Idx() + 1) || CaspalDst0.Idx() & 1) {
|
||||
CaspalDst0 = TMP3;
|
||||
CaspalDst1 = TMP4;
|
||||
}
|
||||
|
||||
// We can't clobber the source, these moves are inherently required due to
|
||||
// ISA limitations. But by making them 64-bit, Firestorm can rename.
|
||||
mov(ARMEmitter::Size::i64Bit, CaspalDst0, Expected0);
|
||||
mov(ARMEmitter::Size::i64Bit, CaspalDst1, Expected1);
|
||||
caspal(EmitSize, CaspalDst0, CaspalDst1, Desired0, Desired1, MemSrc);
|
||||
|
||||
if (CaspalDst0 != Dst0) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst0, CaspalDst0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst1, CaspalDst1);
|
||||
}
|
||||
} else {
|
||||
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
|
||||
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
|
||||
@@ -43,19 +67,19 @@ DEF_OP(CASPair) {
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected.first);
|
||||
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, TMP2, Expected0);
|
||||
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxp(EmitSize, TMP2, Desired.first, Desired.second, MemSrc);
|
||||
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
mov(EmitSize, Dst.first, Expected.first);
|
||||
mov(EmitSize, Dst.second, Expected.second);
|
||||
mov(EmitSize, Dst0, Expected0);
|
||||
mov(EmitSize, Dst1, Expected1);
|
||||
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst.first, TMP2.R());
|
||||
mov(EmitSize, Dst.second, TMP3.R());
|
||||
mov(EmitSize, Dst0, TMP2.R());
|
||||
mov(EmitSize, Dst1, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
|
||||
@@ -11,11 +11,11 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -78,8 +78,12 @@ DEF_OP(ExitFunction) {
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
@@ -112,20 +116,27 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ || Op->Cond.Val == FEXCore::IR::COND_NEQ, "CondJump: Expected simple "
|
||||
"condition");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
|
||||
}
|
||||
|
||||
// TODO: Wire up tbz/tbnz
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
@@ -255,15 +266,13 @@ DEF_OP(InlineSyscall) {
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg == ARMEmitter::Reg::r8) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
} else if (Reg == ARMEmitter::Reg::r4) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
|
||||
} else if (Reg == ARMEmitter::Reg::r5) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RCX]));
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
@@ -427,10 +436,11 @@ DEF_OP(CPUID) {
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
|
||||
// Results want to be 4xi32 scalars
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX.ID()), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX.ID()), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP2, 32, 32);
|
||||
}
|
||||
|
||||
DEF_OP(XGetBV) {
|
||||
@@ -459,11 +469,9 @@ DEF_OP(XGetBV) {
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP1, 32, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -5,7 +5,6 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -17,6 +16,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
@@ -112,6 +112,8 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -204,6 +206,7 @@ DEF_OP(Vector_SToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -236,6 +239,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -266,6 +270,8 @@ DEF_OP(Vector_FToS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -295,6 +301,8 @@ DEF_OP(Vector_FToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -396,6 +404,7 @@ DEF_OP(Vector_FToI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -456,6 +465,7 @@ DEF_OP(Vector_F64ToI32) {
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -654,12 +654,6 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
@@ -724,6 +718,16 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
|
||||
@@ -14,9 +14,6 @@ $end_info$
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -40,25 +37,10 @@ public:
|
||||
explicit Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::string GetName() override {
|
||||
return "JIT";
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
|
||||
[[nodiscard]]
|
||||
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool NeedsOpDispatch() override {
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
void ClearRelocations() override {
|
||||
@@ -67,8 +49,6 @@ public:
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
@@ -114,15 +94,6 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return std::make_pair(GeneralRegisters[Reg.Reg], GeneralRegisters[Reg.Reg + 1]);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
@@ -253,8 +224,6 @@ private:
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
@@ -348,7 +317,8 @@ private:
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
void VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2, ARMEmitter::VRegister Addend);
|
||||
ARMEmitter::VRegister Upper, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2,
|
||||
ARMEmitter::VRegister Addend);
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -47,6 +48,31 @@ DEF_OP(LoadContext) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), STATE, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.X(), Dst2.X(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), STATE, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.D(), Dst2.D(), STATE, Op->Offset); break;
|
||||
case 16: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.Q(), Dst2.Q(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -79,29 +105,46 @@ DEF_OP(StoreContext) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
DEF_OP(StoreContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
unsigned Reg = Op->Reg == Core::CPUState::PF_AS_GREG ? (StaticRegisters.size() - 2) :
|
||||
Op->Reg == Core::CPUState::AF_AS_GREG ? (StaticRegisters.size() - 1) :
|
||||
Op->Reg;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Src1 = GetZeroableReg(Op->Value1);
|
||||
auto Src2 = GetZeroableReg(Op->Value2);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Reg];
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), STATE, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), STATE, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.D(), Src2.D(), STATE, Op->Offset); break;
|
||||
case 16: stp<ARMEmitter::IndexType::OFFSET>(Src1.Q(), Src2.Q(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContextPair size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Op->Reg];
|
||||
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
if (OpSize == 4) {
|
||||
mov(GetReg(Node).W(), reg.W());
|
||||
} else {
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(OpSize == regSize, "expected sized");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Node);
|
||||
@@ -118,6 +161,22 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadPF) {
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 2];
|
||||
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadAF) {
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 1];
|
||||
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
@@ -154,6 +213,28 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StorePF) {
|
||||
const auto Op = IROp->C<IR::IROp_StorePF>();
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 2];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreAF) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreAF>();
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 1];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -367,30 +448,6 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRPairClass) {
|
||||
const auto Src = GetRegPair(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 8: {
|
||||
if (SlotOffset <= 252 && (SlotOffset & 0b11) == 0) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
if (SlotOffset <= 504 && (SlotOffset & 0b111) == 0) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister(GPRPair) size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
@@ -480,30 +537,6 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRPairClass) {
|
||||
const auto Src = GetRegPair(Node);
|
||||
switch (OpSize) {
|
||||
case 8: {
|
||||
if (SlotOffset <= 252 && (SlotOffset & 0b11) == 0) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
if (SlotOffset <= 504 && (SlotOffset & 0b111) == 0) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister(GPRPair) size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
@@ -635,6 +668,7 @@ DEF_OP(LoadMem) {
|
||||
case 8: ldr(Dst.D(), MemSrc); break;
|
||||
case 16: ldr(Dst.Q(), MemSrc); break;
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Operand = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), Operand);
|
||||
break;
|
||||
@@ -644,6 +678,32 @@ DEF_OP(LoadMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), Addr, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.X(), Dst2.X(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), Addr, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.D(), Dst2.D(), Addr, Op->Offset); break;
|
||||
case 16: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.Q(), Dst2.Q(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -717,13 +777,14 @@ DEF_OP(LoadMemTSO) {
|
||||
case 8: ldr(Dst.D(), MemSrc); break;
|
||||
case 16: ldr(Dst.Q(), MemSrc); break;
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
if (VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
// Half-barrier.
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
@@ -736,9 +797,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VLoadVectorMasked with 256-bit operation");
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -833,9 +892,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreVectorMasked with 256-bit operation");
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -1052,9 +1109,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
/// - AddrBase also doesn't need to exist
|
||||
/// - If the instruction is using 64-bit vector indexing or 32-bit addresses where the top-bit isn't set then this is valid!
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreVectorMasked with 256-bit operation");
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
@@ -1238,7 +1293,7 @@ DEF_OP(VLoadVectorElement) {
|
||||
}
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
@@ -1257,7 +1312,7 @@ DEF_OP(VStoreVectorElement) {
|
||||
"size");
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
|
||||
@@ -1280,6 +1335,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1288,11 +1344,6 @@ DEF_OP(VBroadcastFromMem) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid element "
|
||||
"size");
|
||||
|
||||
if (Is256Bit && !HostSupportsSVE256) {
|
||||
LOGMAN_MSG_A_FMT("{}: 256-bit vectors must support SVE256", __func__);
|
||||
return;
|
||||
}
|
||||
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
|
||||
@@ -1319,7 +1370,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
}
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
@@ -1409,6 +1460,37 @@ DEF_OP(Push) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Pop) {
|
||||
const auto Op = IROp->C<IR::IROp_Pop>();
|
||||
const auto Addr = GetReg(Op->InoutAddr.ID());
|
||||
const auto Dst = GetReg(Op->OutValue.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(Dst != Addr, "Invalid");
|
||||
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
ldrb<ARMEmitter::IndexType::POST>(Dst.W(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ldrh<ARMEmitter::IndexType::POST>(Dst.W(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ldr<ARMEmitter::IndexType::POST>(Dst.W(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ldr<ARMEmitter::IndexType::POST>(Dst.X(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Op->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1450,6 +1532,7 @@ DEF_OP(StoreMem) {
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemSrc);
|
||||
break;
|
||||
@@ -1459,6 +1542,32 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetReg(Op->Value1.ID());
|
||||
const auto Src2 = GetReg(Op->Value2.ID());
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), Addr, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), Addr, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.D(), Src2.D(), Addr, Op->Offset); break;
|
||||
case 16: stp<ARMEmitter::IndexType::OFFSET>(Src1.Q(), Src2.Q(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemPair size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1508,7 +1617,7 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
// Half-Barrier.
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
@@ -1521,6 +1630,7 @@ DEF_OP(StoreMemTSO) {
|
||||
case 8: str(Src.D(), MemSrc); break;
|
||||
case 16: str(Src.Q(), MemSrc); break;
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Operand = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, Operand);
|
||||
break;
|
||||
@@ -1540,7 +1650,7 @@ DEF_OP(MemSet) {
|
||||
// that the value is zero, we can optimize any operation larger than 8-bit down to 8-bit to use the MOPS implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
|
||||
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
@@ -1730,7 +1840,7 @@ DEF_OP(MemCpy) {
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemRegDest = GetReg(Op->Dest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->Src.ID());
|
||||
@@ -1743,7 +1853,8 @@ DEF_OP(MemCpy) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Dst0 = GetReg(Op->OutDstAddress.ID());
|
||||
auto Dst1 = GetReg(Op->OutSrcAddress.ID());
|
||||
// If Direction > 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
@@ -1940,40 +2051,40 @@ DEF_OP(MemCpy) {
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
add(Dst.first.X(), TMP1, TMP3);
|
||||
add(Dst.second.X(), TMP2, TMP3);
|
||||
add(Dst0.X(), TMP1, TMP3);
|
||||
add(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
sub(Dst.first.X(), TMP1, TMP3);
|
||||
sub(Dst.second.X(), TMP2, TMP3);
|
||||
sub(Dst0.X(), TMP1, TMP3);
|
||||
sub(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
@@ -2070,6 +2181,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
@@ -2146,6 +2258,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
@@ -2157,6 +2270,11 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::SY);
|
||||
return;
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
@@ -2181,6 +2299,11 @@ DEF_OP(CacheLineClear) {
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClean) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::ST);
|
||||
return;
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClean>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
@@ -2262,6 +2385,7 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
@@ -2269,7 +2393,6 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreNonTemporal with 256-bit operation");
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
const auto OffsetScaled = Offset / 32;
|
||||
stnt1b(Value.Z(), GoverningPredicate, MemReg, OffsetScaled);
|
||||
@@ -2304,6 +2427,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2311,7 +2435,6 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreNonTemporal with 256-bit operation");
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
const auto OffsetScaled = Offset / 32;
|
||||
ldnt1b(Dst.Z(), GoverningPredicate, MemReg, OffsetScaled);
|
||||
|
||||
@@ -18,6 +18,10 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AllocateGPR) {}
|
||||
DEF_OP(AllocateGPRAfter) {}
|
||||
DEF_OP(AllocateFPR) {}
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
@@ -254,18 +258,7 @@ DEF_OP(ProcessorID) {
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDRRS);
|
||||
} else {
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDR);
|
||||
}
|
||||
|
||||
// If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100
|
||||
cset(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Condition::CC_NE);
|
||||
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
|
||||
@@ -9,47 +9,22 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Pair = GetRegPair(Op->Pair.ID());
|
||||
const auto Src = Op->Element == 0 ? Pair.first : Pair.second;
|
||||
|
||||
if (Dst != Src) {
|
||||
mov(ConvertSize48(IROp), Dst, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Invalid size");
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> Dst = GetRegPair(Node);
|
||||
ARMEmitter::Register RegFirst = GetReg(Op->Lower.ID());
|
||||
ARMEmitter::Register RegSecond = GetReg(Op->Upper.ID());
|
||||
ARMEmitter::Register RegTmp = TMP1.R();
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (Dst.first.Idx() != RegSecond.Idx()) {
|
||||
mov(EmitSize, Dst.first, RegFirst);
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
} else if (Dst.second.Idx() != RegFirst.Idx()) {
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
mov(EmitSize, Dst.first, RegFirst);
|
||||
} else {
|
||||
mov(EmitSize, RegTmp, RegFirst);
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
mov(EmitSize, Dst.first, RegTmp);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Copy) {
|
||||
auto Op = IROp->C<IR::IROp_Copy>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(RMWHandle) {
|
||||
auto Op = IROp->C<IR::IROp_RMWHandle>();
|
||||
auto Dest = GetReg(Node);
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Dest != Src) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dest, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A.ID()), B = GetReg(Op->B.ID());
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
@@ -34,7 +35,8 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// Allocate a region of memory that we can use to back our block pointers
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize));
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
// We need 32KB per guest page (One pointer per byte)
|
||||
@@ -52,8 +54,8 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
const size_t TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
ctx->SyscallHandler->UnmarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// No need to free BlockLinks map.
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
@@ -63,7 +65,7 @@ void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
@@ -71,7 +73,7 @@ void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -117,6 +118,7 @@ public:
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
PossiblySetNZCVBits = ~0U;
|
||||
CFInverted = CFInvertedABI;
|
||||
FlushRegisterCache();
|
||||
|
||||
// New block needs to reset segment telemetry.
|
||||
@@ -149,11 +151,12 @@ public:
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
|
||||
FlushRegisterCache();
|
||||
|
||||
// The jump will ignore the sources, so it doesn't matter what we put here.
|
||||
// Put an inline constant so RA+codegen will ignore altogether.
|
||||
auto Placeholder = _InlineConstant(0);
|
||||
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
|
||||
return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, 0, true);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
|
||||
FlushRegisterCache();
|
||||
auto InlineConst = _InlineConstant(Bit);
|
||||
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, 0, false);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
|
||||
FlushRegisterCache();
|
||||
@@ -294,8 +297,7 @@ public:
|
||||
};
|
||||
|
||||
void UnhandledOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void MOVGPROp(OpcodeArgs);
|
||||
void MOVGPROp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
@@ -310,20 +312,16 @@ public:
|
||||
void IRETOp(OpcodeArgs);
|
||||
void CallbackReturnOp(OpcodeArgs);
|
||||
void SecondaryALUOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void ADCOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void SBBOp(OpcodeArgs);
|
||||
void ADCOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void SBBOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void SALCOp(OpcodeArgs);
|
||||
void PUSHOp(OpcodeArgs);
|
||||
void PUSHREGOp(OpcodeArgs);
|
||||
void PUSHAOp(OpcodeArgs);
|
||||
template<uint32_t SegmentReg>
|
||||
void PUSHSegmentOp(OpcodeArgs);
|
||||
void PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg);
|
||||
void POPOp(OpcodeArgs);
|
||||
void POPAOp(OpcodeArgs);
|
||||
template<uint32_t SegmentReg>
|
||||
void POPSegmentOp(OpcodeArgs);
|
||||
void POPSegmentOp(OpcodeArgs, uint32_t SegmentReg);
|
||||
void LEAVEOp(OpcodeArgs);
|
||||
void CALLOp(OpcodeArgs);
|
||||
void CALLAbsoluteOp(OpcodeArgs);
|
||||
@@ -332,21 +330,18 @@ public:
|
||||
void LoopOp(OpcodeArgs);
|
||||
void JUMPOp(OpcodeArgs);
|
||||
void JUMPAbsoluteOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void TESTOp(OpcodeArgs);
|
||||
void TESTOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void MOVSXDOp(OpcodeArgs);
|
||||
void MOVSXOp(OpcodeArgs);
|
||||
void MOVZXOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void CMPOp(OpcodeArgs);
|
||||
void CMPOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void SETccOp(OpcodeArgs);
|
||||
void CQOOp(OpcodeArgs);
|
||||
void CDQOp(OpcodeArgs);
|
||||
void XCHGOp(OpcodeArgs);
|
||||
void SAHFOp(OpcodeArgs);
|
||||
void LAHFOp(OpcodeArgs);
|
||||
template<bool ToSeg>
|
||||
void MOVSegOp(OpcodeArgs);
|
||||
void MOVSegOp(OpcodeArgs, bool ToSeg);
|
||||
void FLAGControlOp(OpcodeArgs);
|
||||
void MOVOffsetOp(OpcodeArgs);
|
||||
void CMOVOp(OpcodeArgs);
|
||||
@@ -354,19 +349,14 @@ public:
|
||||
void XGetBVOp(OpcodeArgs);
|
||||
uint32_t LoadConstantShift(X86Tables::DecodedOp Op, bool Is1Bit);
|
||||
void SHLOp(OpcodeArgs);
|
||||
template<bool SHL1Bit>
|
||||
void SHLImmediateOp(OpcodeArgs);
|
||||
void SHLImmediateOp(OpcodeArgs, bool SHL1Bit);
|
||||
void SHROp(OpcodeArgs);
|
||||
template<bool SHR1Bit>
|
||||
void SHRImmediateOp(OpcodeArgs);
|
||||
void SHRImmediateOp(OpcodeArgs, bool SHR1Bit);
|
||||
void SHLDOp(OpcodeArgs);
|
||||
void SHLDImmediateOp(OpcodeArgs);
|
||||
void SHRDOp(OpcodeArgs);
|
||||
void SHRDImmediateOp(OpcodeArgs);
|
||||
template<bool IsImmediate, bool Is1Bit>
|
||||
void ASHROp(OpcodeArgs);
|
||||
template<bool Left, bool IsImmediate, bool Is1Bit>
|
||||
void RotateOp(OpcodeArgs);
|
||||
void ASHROp(OpcodeArgs, bool IsImmediate, bool Is1Bit);
|
||||
void RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool Is1Bit);
|
||||
void RCROp1Bit(OpcodeArgs);
|
||||
void RCROp8x1Bit(OpcodeArgs);
|
||||
@@ -376,8 +366,6 @@ public:
|
||||
void RCLOp(OpcodeArgs);
|
||||
void RCLSmallerOp(OpcodeArgs);
|
||||
|
||||
template<uint32_t SrcIndex, enum BTAction Action>
|
||||
void BTOp(OpcodeArgs);
|
||||
void BTOp(OpcodeArgs, uint32_t SrcIndex, enum BTAction Action);
|
||||
|
||||
void IMUL1SrcOp(OpcodeArgs);
|
||||
@@ -425,10 +413,8 @@ public:
|
||||
FS,
|
||||
GS,
|
||||
};
|
||||
template<Segment Seg>
|
||||
void ReadSegmentReg(OpcodeArgs);
|
||||
template<Segment Seg>
|
||||
void WriteSegmentReg(OpcodeArgs);
|
||||
void ReadSegmentReg(OpcodeArgs, Segment Seg);
|
||||
void WriteSegmentReg(OpcodeArgs, Segment Seg);
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
@@ -454,32 +440,22 @@ public:
|
||||
|
||||
void MOVQOp(OpcodeArgs, VectorOpType VectorType);
|
||||
void MOVQMMXOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void MOVMSKOp(OpcodeArgs);
|
||||
void MOVMSKOp(OpcodeArgs, size_t ElementSize);
|
||||
void MOVMSKOpOne(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PUNPCKLOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PUNPCKHOp(OpcodeArgs);
|
||||
void PUNPCKLOp(OpcodeArgs, size_t ElementSize);
|
||||
void PUNPCKHOp(OpcodeArgs, size_t ElementSize);
|
||||
void PSHUFBOp(OpcodeArgs);
|
||||
template<bool Low>
|
||||
void PSHUFWOp(OpcodeArgs);
|
||||
void PSHUFWOp(OpcodeArgs, bool Low);
|
||||
void PSHUFW8ByteOp(OpcodeArgs);
|
||||
void PSHUFDOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRLDOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRLI(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSLLI(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSLL(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRAOp(OpcodeArgs);
|
||||
void PSRLDOp(OpcodeArgs, size_t ElementSize);
|
||||
void PSRLI(OpcodeArgs, size_t ElementSize);
|
||||
void PSLLI(OpcodeArgs, size_t ElementSize);
|
||||
void PSLL(OpcodeArgs, size_t ElementSize);
|
||||
void PSRAOp(OpcodeArgs, size_t ElementSize);
|
||||
void PSRLDQ(OpcodeArgs);
|
||||
void PSLLDQ(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRAIOp(OpcodeArgs);
|
||||
void PSRAIOp(OpcodeArgs, size_t ElementSize);
|
||||
void MOVDDUPOp(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void CVTGPR_To_FPR(OpcodeArgs);
|
||||
@@ -501,13 +477,11 @@ public:
|
||||
void LZCNT(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void SHUFOp(OpcodeArgs);
|
||||
void SHUFOp(OpcodeArgs, size_t ElementSize);
|
||||
template<size_t ElementSize>
|
||||
void PINSROp(OpcodeArgs);
|
||||
void InsertPSOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PExtrOp(OpcodeArgs);
|
||||
void PExtrOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void PSIGN(OpcodeArgs);
|
||||
@@ -601,8 +575,7 @@ public:
|
||||
void VPBLENDDOp(OpcodeArgs);
|
||||
void VPBLENDWOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VBROADCASTOp(OpcodeArgs);
|
||||
void VBROADCASTOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VDPPOp(OpcodeArgs);
|
||||
@@ -611,8 +584,7 @@ public:
|
||||
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void VHADDPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void VHSUBPOp(OpcodeArgs);
|
||||
void VHSUBPOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VINSERTOp(OpcodeArgs);
|
||||
void VINSERTPSOp(OpcodeArgs);
|
||||
@@ -635,11 +607,9 @@ public:
|
||||
|
||||
void VMPSADBWOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPACKSSOp(OpcodeArgs);
|
||||
void VPACKSSOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPACKUSOp(OpcodeArgs);
|
||||
void VPACKUSOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VPALIGNROp(OpcodeArgs);
|
||||
|
||||
@@ -656,8 +626,7 @@ public:
|
||||
void VPERMDOp(OpcodeArgs);
|
||||
void VPERMQOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPERMILImmOp(OpcodeArgs);
|
||||
void VPERMILImmOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
Ref VPERMILRegOpImpl(OpSize DstSize, size_t ElementSize, Ref Src, Ref Indices);
|
||||
template<size_t ElementSize>
|
||||
@@ -665,8 +634,7 @@ public:
|
||||
|
||||
void VPHADDSWOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPHSUBOp(OpcodeArgs);
|
||||
void VPHSUBOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPHSUBSWOp(OpcodeArgs);
|
||||
|
||||
void VPINSRBOp(OpcodeArgs);
|
||||
@@ -691,40 +659,30 @@ public:
|
||||
|
||||
void VPSHUFBOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Low>
|
||||
void VPSHUFWOp(OpcodeArgs);
|
||||
void VPSHUFWOp(OpcodeArgs, size_t ElementSize, bool Low);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSLLOp(OpcodeArgs);
|
||||
void VPSLLOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPSLLDQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void VPSLLIOp(OpcodeArgs);
|
||||
void VPSLLIOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPSLLVOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRAOp(OpcodeArgs);
|
||||
void VPSRAOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRAIOp(OpcodeArgs);
|
||||
void VPSRAIOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VPSRAVDOp(OpcodeArgs);
|
||||
void VPSRLVOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRLDOp(OpcodeArgs);
|
||||
void VPSRLDOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPSRLDQOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPUNPCKHOp(OpcodeArgs);
|
||||
void VPUNPCKHOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPUNPCKLOp(OpcodeArgs);
|
||||
void VPUNPCKLOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRLIOp(OpcodeArgs);
|
||||
void VPSRLIOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VSHUFOp(OpcodeArgs);
|
||||
void VSHUFOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VTESTPOp(OpcodeArgs);
|
||||
@@ -890,8 +848,7 @@ public:
|
||||
void RDTSCPOp(OpcodeArgs);
|
||||
void RDPIDOp(OpcodeArgs);
|
||||
|
||||
template<bool ForStore, bool Stream, uint8_t Level>
|
||||
void Prefetch(OpcodeArgs);
|
||||
void Prefetch(OpcodeArgs, bool ForStore, bool Stream, uint8_t Level);
|
||||
|
||||
void PSADBW(OpcodeArgs);
|
||||
|
||||
@@ -937,8 +894,7 @@ public:
|
||||
template<size_t ElementSize>
|
||||
void VectorBlend(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VectorVariableBlend(OpcodeArgs);
|
||||
void VectorVariableBlend(OpcodeArgs, size_t ElementSize);
|
||||
void PTestOpImpl(OpSize Size, Ref Dest, Ref Src);
|
||||
void PTestOp(OpcodeArgs);
|
||||
void PHMINPOSUWOp(OpcodeArgs);
|
||||
@@ -1227,6 +1183,11 @@ public:
|
||||
}
|
||||
|
||||
void FlushRegisterCache(bool SRAOnly = false) {
|
||||
// At block boundaries, fix up the carry flag.
|
||||
if (!SRAOnly) {
|
||||
RectifyCarryInvert(CFInvertedABI);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
@@ -1253,8 +1214,12 @@ public:
|
||||
uint32_t Index = 63 - std::countl_zero(Bits);
|
||||
Ref Value = RegCache.Value[Index];
|
||||
|
||||
if (Index >= GPR0Index && Index <= AFIndex) {
|
||||
if (Index >= GPR0Index && Index <= GPR15Index) {
|
||||
_StoreRegister(Value, Index - GPR0Index, GPRClass, GPRSize);
|
||||
} else if (Index == PFIndex) {
|
||||
_StorePF(Value, GPRSize);
|
||||
} else if (Index == AFIndex) {
|
||||
_StoreAF(Value, GPRSize);
|
||||
} else if (Index >= FPR0Index && Index <= FPR15Index) {
|
||||
_StoreRegister(Value, Index - FPR0Index, FPRClass, VectorSize);
|
||||
} else if (Index == DFIndex) {
|
||||
@@ -1262,7 +1227,23 @@ public:
|
||||
} else {
|
||||
bool Partial = RegCache.Partial & (1ull << Index);
|
||||
unsigned Size = Partial ? 8 : CacheIndexToSize(Index);
|
||||
_StoreContext(Size, CacheIndexClass(Index), Value, CacheIndexToContextOffset(Index));
|
||||
uint64_t NextBit = (1ull << (Index - 1));
|
||||
uint32_t Offset = CacheIndexToContextOffset(Index);
|
||||
auto Class = CacheIndexClass(Index);
|
||||
|
||||
// Use stp where possible to store multiple values at a time. This accelerates AVX.
|
||||
// TODO: this is all really confusing because of backwards iteration,
|
||||
// can we peel back that hack?
|
||||
if ((Bits & NextBit) && !Partial && Size >= 4 && CacheIndexToContextOffset(Index - 1) == Offset - Size && (Offset - Size) / Size < 64) {
|
||||
LOGMAN_THROW_A_FMT(CacheIndexClass(Index - 1) == Class, "construction");
|
||||
LOGMAN_THROW_A_FMT((Offset % Size) == 0, "construction");
|
||||
Ref ValueNext = RegCache.Value[Index - 1];
|
||||
|
||||
_StoreContextPair(Size, Class, ValueNext, Value, Offset - Size);
|
||||
Bits &= ~NextBit;
|
||||
} else {
|
||||
_StoreContext(Size, Class, Value, Offset);
|
||||
}
|
||||
}
|
||||
|
||||
Bits &= ~(1ull << Index);
|
||||
@@ -1345,6 +1326,18 @@ private:
|
||||
bool NZCVDirty {};
|
||||
uint32_t PossiblySetNZCVBits {};
|
||||
|
||||
// Set if the host carry is inverted from the guest carry. This is set after
|
||||
// subtraction, because arm64 and x86 have inverted borrow flags, but clear
|
||||
// after addition.
|
||||
//
|
||||
// All CF access needs to maintain this flag. cfinv may be inserted at the end
|
||||
// of a block to rectify to the FEX convention (current convention: NOT
|
||||
// INVERTED).
|
||||
bool CFInverted {};
|
||||
|
||||
// FEX convention for CF at the end of blocks: INVERTED.
|
||||
const bool CFInvertedABI {true};
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
bool HandledLock {false};
|
||||
bool DecodeFailure {false};
|
||||
@@ -1365,8 +1358,7 @@ private:
|
||||
void AVXVectorALUOp(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOp(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVXVectorVariableBlend(OpcodeArgs);
|
||||
void AVXVectorVariableBlend(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void AVXVariableShiftImpl(OpcodeArgs, IROps IROp);
|
||||
|
||||
@@ -1436,7 +1428,7 @@ private:
|
||||
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
Ref VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
Ref VFCMPOpImpl(OpSize Size, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpSize SrcSize, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1582,23 +1574,6 @@ private:
|
||||
return IR::SizeToOpSize(GetSrcSize(Op));
|
||||
}
|
||||
|
||||
static inline constexpr unsigned NZCVIndexMask(unsigned BitMask) {
|
||||
unsigned NZCVMask {};
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
return NZCVMask;
|
||||
}
|
||||
|
||||
// Set flag tracking to prepare for an operation that directly writes NZCV. If
|
||||
// some bits are known to be zeroed, the PossiblySetNZCVBits mask can be
|
||||
// passed. Otherwise, it defaults to assuming all bits may be set after
|
||||
@@ -1624,6 +1599,10 @@ private:
|
||||
// Special case of the above where we are known to zero C/V
|
||||
void HandleNZ00Write() {
|
||||
HandleNZCVWrite((1u << 31) | (1u << 30));
|
||||
|
||||
// Host carry will be implicitly zeroed, and we want guest carry zeroed as
|
||||
// well. So do not invert.
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
Ref GetNZCV() {
|
||||
@@ -1645,9 +1624,34 @@ private:
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, Ref Res) {
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, Ref Res, bool SetPF = false) {
|
||||
HandleNZ00Write();
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
|
||||
// x - 0 = x. NZ set according to Res. C always set. V always unset. This
|
||||
// matches what we want since we want carry inverted.
|
||||
//
|
||||
// This is currently worse for 8/16-bit, but that should be optimized. TODO
|
||||
if (SrcSize >= 4) {
|
||||
if (SetPF) {
|
||||
CalculatePF(_SubWithFlags(IR::SizeToOpSize(SrcSize), Res, _Constant(0)));
|
||||
} else {
|
||||
_SubNZCV(IR::SizeToOpSize(SrcSize), Res, _Constant(0));
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits |= 1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
CFInverted = true;
|
||||
} else {
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
CFInverted = false;
|
||||
|
||||
if (SetPF) {
|
||||
CalculatePF(Res);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void SetNZP_ZeroCV(unsigned SrcSize, Ref Res) {
|
||||
SetNZ_ZeroCV(SrcSize, Res, true);
|
||||
}
|
||||
|
||||
void InsertNZCV(unsigned BitOffset, Ref Value, signed FlagOffset, bool MustMask) {
|
||||
@@ -1692,19 +1696,30 @@ private:
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
}
|
||||
|
||||
void CarryInvert() {
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
// Ensure the carry invert flag matches the desired form. Used before an
|
||||
// operation reading carry or at the end of a block.
|
||||
void RectifyCarryInvert(bool RequiredInvert) {
|
||||
if (CFInverted != RequiredInvert) {
|
||||
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
|
||||
// Invert as NZCV.
|
||||
_CarryInvert();
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
|
||||
// Invert as NZCV.
|
||||
_CarryInvert();
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
|
||||
CFInverted ^= true;
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits |= 1u << Bit;
|
||||
LOGMAN_THROW_AA_FMT(CFInverted == RequiredInvert, "post condition");
|
||||
}
|
||||
|
||||
void CarryInvert() {
|
||||
CFInverted ^= true;
|
||||
PossiblySetNZCVBits |= 1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
@@ -1712,24 +1727,36 @@ private:
|
||||
SetRFLAG(Value, BitOffset, ValueOffset, MustMask);
|
||||
}
|
||||
|
||||
void SetCFDirect(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
Value = _Xor(OpSize::i64Bit, Value, _InlineConstant(1ull << ValueOffset));
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void SetCFInverted(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void SetRFLAG(Ref Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
StoreRegister(Core::CPUState::PF_AS_GREG, false, Value);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
StoreRegister(Core::CPUState::AF_AS_GREG, false, Value);
|
||||
} else {
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1756,14 +1783,10 @@ private:
|
||||
|
||||
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClassType {COND_PL} : CondClassType {COND_MI};
|
||||
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClassType {COND_NEQ} : CondClassType {COND_EQ};
|
||||
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClassType {COND_ULT} : CondClassType {COND_UGE};
|
||||
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClassType {COND_FNU} : CondClassType {COND_FU};
|
||||
|
||||
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
|
||||
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
|
||||
case X86State::RFLAG_CF_RAW_LOC: return {Invert ? COND_ULT : COND_UGE};
|
||||
case X86State::RFLAG_OF_RAW_LOC: return {Invert ? COND_FNU : COND_FU};
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
@@ -1855,6 +1878,10 @@ private:
|
||||
if (Size == 8) {
|
||||
RegCache.Partial |= Bit;
|
||||
}
|
||||
} else if (Index == PFIndex) {
|
||||
RegCache.Value[Index] = _LoadPF(Size);
|
||||
} else if (Index == AFIndex) {
|
||||
RegCache.Value[Index] = _LoadAF(Size);
|
||||
} else {
|
||||
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
|
||||
}
|
||||
@@ -1865,6 +1892,43 @@ private:
|
||||
return RegCache.Value[Index];
|
||||
}
|
||||
|
||||
RefPair AllocatePair(FEXCore::IR::RegisterClassType Class, uint8_t Size) {
|
||||
if (Class == FPRClass) {
|
||||
return {_AllocateFPR(Size, Size), _AllocateFPR(Size, Size)};
|
||||
} else {
|
||||
return {_AllocateGPR(false), _AllocateGPR(false)};
|
||||
}
|
||||
}
|
||||
|
||||
RefPair LoadContextPair_Uncached(FEXCore::IR::RegisterClassType Class, uint8_t Size, unsigned Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadContextPair(Size, Class, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, uint8_t Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index != DFIndex, "must be pairable");
|
||||
|
||||
// Try to load a pair into the cache
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
if (((RegCache.Partial | RegCache.Cached) & Bits) == 0 && ((Offset / Size) < 64)) {
|
||||
auto Values = LoadContextPair_Uncached(RegClass, Size, Offset);
|
||||
RegCache.Value[Index] = Values.Low;
|
||||
RegCache.Value[Index + 1] = Values.High;
|
||||
RegCache.Cached |= Bits;
|
||||
if (Size == 8) {
|
||||
RegCache.Partial |= Bits;
|
||||
}
|
||||
return Values;
|
||||
}
|
||||
|
||||
// Fallback on a pair of loads
|
||||
return {
|
||||
.Low = LoadRegCache(Offset, Index, RegClass, Size),
|
||||
.High = LoadRegCache(Offset + Size, Index + 1, RegClass, Size),
|
||||
};
|
||||
}
|
||||
|
||||
Ref LoadGPR(uint8_t Reg) {
|
||||
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, CTX->GetGPRSize());
|
||||
}
|
||||
@@ -1873,6 +1937,10 @@ private:
|
||||
return LoadRegCache(CacheIndexToContextOffset(Index), Index, CacheIndexClass(Index), Size);
|
||||
}
|
||||
|
||||
RefPair LoadContextPair(uint8_t Size, uint8_t Index) {
|
||||
return LoadRegCachePair(CacheIndexToContextOffset(Index), Index, CacheIndexClass(Index), Size);
|
||||
}
|
||||
|
||||
Ref LoadContext(uint8_t Index) {
|
||||
return LoadContext(CacheIndexToSize(Index), Index);
|
||||
}
|
||||
@@ -1912,6 +1980,12 @@ private:
|
||||
|
||||
Ref GetRFLAG(unsigned BitOffset, bool Invert = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
// Handle the CFInverted state internally so GetRFLAG is safe regardless
|
||||
// of the invert state. This simplifies the call sites.
|
||||
if (BitOffset == X86State::RFLAG_CF_RAW_LOC) {
|
||||
Invert ^= CFInverted;
|
||||
}
|
||||
|
||||
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
|
||||
return _Constant(Invert ? 1 : 0);
|
||||
} else if (NZCVDirty) {
|
||||
@@ -1923,6 +1997,8 @@ private:
|
||||
return Value;
|
||||
}
|
||||
} else {
|
||||
// Because we explicitly inverted for CF above, we use the unsafe
|
||||
// _NZCVSelect rather than the safe CF-aware version.
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), _Constant(1), _Constant(0));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
@@ -1956,57 +2032,55 @@ private:
|
||||
return _AddShift(OpSize::i64Bit, X, LoadDF(), ShiftType::LSL, Shift);
|
||||
}
|
||||
|
||||
// Safe version of NZCVSelect that handles inverted carries automatically.
|
||||
Ref NZCVSelect(OpSize OpSize, CondClassType Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
|
||||
switch (Cond) {
|
||||
case IR::COND_UGE: /* cs */
|
||||
case IR::COND_ULT: /* cc */
|
||||
// Invert the condition to match our expectations.
|
||||
if (CarryIsInverted != CFInverted) {
|
||||
Cond = {Cond == COND_UGE ? COND_ULT : COND_UGE};
|
||||
}
|
||||
break;
|
||||
|
||||
case IR::COND_UGT: /* hi */
|
||||
case IR::COND_ULE: /* ls */
|
||||
// No clever optimization we can do here, rectify carry itself.
|
||||
RectifyCarryInvert(CarryIsInverted);
|
||||
break;
|
||||
|
||||
default:
|
||||
// No other condition codes read carry so no need to rectify.
|
||||
break;
|
||||
}
|
||||
|
||||
return _NZCVSelect(OpSize, Cond, TrueV, FalseV);
|
||||
}
|
||||
|
||||
// Compares two floats and sets flags for a COMISS instruction
|
||||
void Comiss(size_t ElementSize, Ref Src1, Ref Src2, bool InvalidateAF = false) {
|
||||
// First, set flags according to Arm FCMP.
|
||||
HandleNZCVWrite();
|
||||
_FCmp(ElementSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
ComissFlags(InvalidateAF);
|
||||
}
|
||||
|
||||
// Sets flags for a COMISS instruction
|
||||
void ComissFlags(bool InvalidateAF = false) {
|
||||
// Now set COMISS flags by converts NZCV from the Arm representation to an
|
||||
// eXternal representation that's totally not a euphemism for x86, nuh-uh.
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
|
||||
// after. We can recover "unordered" after axflag as (Z && !C), but
|
||||
// there's no condition code for this so it would take 2 instructions
|
||||
// instead of one, which seems worse than doing 1 op before and breaking
|
||||
// the fusion.
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
Ref PFInvert = _NZCVSelect(OpSize::i32Bit, CondClassType {COND_FNU}, _Constant(1), _Constant(0));
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PFInvert);
|
||||
|
||||
// For the rest, this one weird a64 instruction maps exactly to what x86
|
||||
// needs. What a coincidence!
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// It does assume we invert CF internally, which is still TODO for us. For
|
||||
// now, add a cfinv to deal. Hopefully we delete this later.
|
||||
CarryInvert();
|
||||
} else {
|
||||
Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
Ref C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
Ref V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
// We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do
|
||||
// this all with shifted-or's on non-flagm platforms.
|
||||
ZeroNZCV();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Or(OpSize::i32Bit, C_inv, V));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
|
||||
// Note that we store PF inverted.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Xor(OpSize::i32Bit, V, _Constant(1)));
|
||||
}
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
|
||||
// after. We can recover "unordered" after axflag as (Z && !C), but
|
||||
// there's no condition code for this so it would take 2 instructions
|
||||
// instead of one, which seems worse than doing 1 op before and breaking
|
||||
// the fusion.
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
Ref V_inv = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(V_inv);
|
||||
|
||||
if (!InvalidateAF) {
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
@@ -2014,6 +2088,15 @@ private:
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Convert NZCV from the Arm representation to an eXternal representation
|
||||
// that's totally not a euphemism for x86, nuh-uh. But maps to exactly we
|
||||
// need, what a coincidence!
|
||||
//
|
||||
// Our AXFlag emulation on FlagM2-less systems needs V_inv passed.
|
||||
_AXFlag(CTX->HostFeatures.SupportsFlagM2 ? Invalid() : V_inv);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
|
||||
@@ -2025,11 +2108,12 @@ private:
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// Convert to x86 flags, saves us from or'ing after.
|
||||
_AXFlag();
|
||||
_AXFlag(Invalid());
|
||||
PossiblySetNZCVBits = ~0;
|
||||
CFInverted = true;
|
||||
|
||||
// Copy the values. CF is inverted from the axflag result, ZF is as-is.
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
// Copy the values.
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
} else {
|
||||
Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
@@ -2050,18 +2134,10 @@ private:
|
||||
auto OldPF = GetRFLAG(X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
HandleNZCV_RMW();
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF));
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF, CFInverted));
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
std::pair<Ref, Ref> ExtractPair(OpSize Size, Ref Pair) {
|
||||
// Extract high first. This is a hack to improve coalescing.
|
||||
Ref Hi = _ExtractElementPair(Size, Pair, 1);
|
||||
Ref Lo = _ExtractElementPair(Size, Pair, 0);
|
||||
|
||||
return std::make_pair(Lo, Hi);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
// replaced with NewOp. Useful for generic building code. Not safe in general.
|
||||
// but does the right handling of ImplicitFlagClobber at least and must be
|
||||
@@ -2132,7 +2208,7 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
|
||||
@@ -2218,7 +2294,8 @@ private:
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
Ref LoadPFRaw(bool Invert);
|
||||
Ref LoadPFRaw(bool Mask, bool Invert);
|
||||
Ref SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref LoadAF();
|
||||
void FixupAF();
|
||||
void SetAFAndFixup(Ref AF);
|
||||
@@ -2226,6 +2303,8 @@ private:
|
||||
void CalculatePF(Ref Res);
|
||||
void CalculateAF(Ref Src1, Ref Src2);
|
||||
|
||||
Ref IncrementByCarry(OpSize OpSize, Ref Src);
|
||||
|
||||
void CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub);
|
||||
Ref CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2);
|
||||
Ref CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2);
|
||||
@@ -2241,14 +2320,7 @@ private:
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_BEXTR(Ref Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, Ref Src);
|
||||
void CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Res, Ref Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, Ref Res, Ref Src);
|
||||
void CalculateFlags_POPCOUNT(Ref Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src);
|
||||
void CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result);
|
||||
void CalculateFlags_RDRAND(Ref Src);
|
||||
/** @} */
|
||||
|
||||
Ref AndConst(FEXCore::IR::OpSize Size, Ref Node, uint64_t Const) {
|
||||
@@ -2285,8 +2357,16 @@ private:
|
||||
uint64_t Entry;
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) {
|
||||
if (Class == FPRClass) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
if (CTX->IsAtomicTSOEnabled()) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
@@ -2294,7 +2374,7 @@ private:
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
if (CTX->IsAtomicTSOEnabled()) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
@@ -2302,7 +2382,7 @@ private:
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO;
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
@@ -2312,8 +2392,46 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, uint8_t Size) {
|
||||
AddressMode Out {};
|
||||
|
||||
signed OffsetEl = A.Offset / Size;
|
||||
if ((A.Offset % Size) == 0 && OffsetEl >= -64 && OffsetEl < 64) {
|
||||
Out.Offset = A.Offset;
|
||||
A.Offset = 0;
|
||||
}
|
||||
|
||||
Out.Base = LoadEffectiveAddress(A, true, false);
|
||||
return Out;
|
||||
}
|
||||
|
||||
|
||||
RefPair LoadMemPair(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Base, unsigned Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadMemPair(Class, Size, Base, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
|
||||
RefPair _LoadMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use ldp if possible, otherwise fallback on two loads.
|
||||
if (!AtomicTSO && !A.Segment && Size >= 4 & Size <= 16) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, A.Base, A.Offset);
|
||||
} else {
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
|
||||
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value, uint8_t Align = 1) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO;
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
@@ -2323,8 +2441,41 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
Ref Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, Ref ssa0) {
|
||||
return _Prefetch(ForStore, Stream, CacheLevel, ssa0, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
void _StoreMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value1, Ref Value2, uint8_t Align = 1) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use stp if possible, otherwise fallback on two stores.
|
||||
if (!AtomicTSO && !A.Segment && Size >= 4 & Size <= 16) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, A.Base, A.Offset);
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, Size, A, Value1, 1);
|
||||
A.Offset += Size;
|
||||
_StoreMemAutoTSO(Class, Size, A, Value2, 1);
|
||||
}
|
||||
}
|
||||
|
||||
Ref Pop(uint8_t Size, Ref SP_RMW) {
|
||||
Ref Value = _AllocateGPR(false);
|
||||
_Pop(Size, SP_RMW, Value);
|
||||
return Value;
|
||||
}
|
||||
|
||||
Ref Pop(uint8_t Size) {
|
||||
Ref SP = _RMWHandle(LoadGPRRegister(X86State::REG_RSP));
|
||||
Ref Value = _AllocateGPR(false);
|
||||
|
||||
_Pop(Size, SP, Value);
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, SP);
|
||||
return Value;
|
||||
}
|
||||
|
||||
void Push(uint8_t Size, Ref Value) {
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
auto NewSP = _Push(CTX->GetGPRSize(), Size, Value, OldSP);
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
}
|
||||
|
||||
void InstallHostSpecificOpcodeHandlers();
|
||||
@@ -2335,6 +2486,23 @@ private:
|
||||
void CheckLegacySegmentRead(Ref NewNode, uint32_t SegmentReg);
|
||||
};
|
||||
|
||||
constexpr inline void InstallToTable(auto& FinalTable, const auto& LocalTable) {
|
||||
for (const auto& Op : LocalTable) {
|
||||
auto OpNum = std::get<0>(Op);
|
||||
auto Dispatcher = std::get<2>(Op);
|
||||
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
|
||||
auto& TableOp = FinalTable[OpNum + i];
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
if (TableOp.OpcodeDispatcher) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}", TableOp.Name);
|
||||
}
|
||||
#endif
|
||||
|
||||
TableOp.OpcodeDispatcher = Dispatcher;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void InstallOpcodeHandlers(Context::OperatingMode Mode);
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -462,17 +462,6 @@ void OpDispatchBuilder::InstallAVX128Handlers() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
auto InstallToTable = [](auto& FinalTable, auto& LocalTable) {
|
||||
for (auto Op : LocalTable) {
|
||||
auto OpNum = std::get<0>(Op);
|
||||
auto Dispatcher = std::get<2>(Op);
|
||||
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
|
||||
LOGMAN_THROW_A_FMT(FinalTable[OpNum + i].OpcodeDispatcher == nullptr, "Duplicate Entry");
|
||||
FinalTable[OpNum + i].OpcodeDispatcher = Dispatcher;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableOps, AVX128Table);
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableGroupOps, VEX128TableGroupOps);
|
||||
if (CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
@@ -508,10 +497,11 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
LOGMAN_THROW_AA_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(FPRClass, 16, A, 1),
|
||||
.High = NeedsHigh ? _LoadMemAutoTSO(FPRClass, 16, HighA, 1) : nullptr,
|
||||
};
|
||||
if (NeedsHigh) {
|
||||
return _LoadMemPairAutoTSO(FPRClass, 16, A, 1);
|
||||
} else {
|
||||
return {.Low = _LoadMemAutoTSO(FPRClass, 16, A, 1)};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -557,13 +547,10 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
} else {
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
_StoreMemAutoTSO(FPRClass, 16, A, Src.Low, 1);
|
||||
|
||||
if (Src.High) {
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
_StoreMemAutoTSO(FPRClass, 16, HighA, Src.High, 1);
|
||||
_StoreMemPairAutoTSO(FPRClass, 16, A, Src.Low, Src.High, 1);
|
||||
} else {
|
||||
_StoreMemAutoTSO(FPRClass, 16, A, Src.Low, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -870,7 +857,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) {
|
||||
///< VMOVLPS/PD xmm1, xmm2, mem64
|
||||
// Bits[63:0] come from Src2[63:0]
|
||||
// Bits[127:64] come from Src1[127:64]
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], OpSize::i64Bit, Op->Flags, {.LoadData = false});
|
||||
auto Src2 = MakeSegmentAddress(Op, Op->Src[1]);
|
||||
Ref Result_Low = _VLoadVectorElement(OpSize::i128Bit, OpSize::i64Bit, Src1.Low, 0, Src2);
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i128Bit);
|
||||
|
||||
@@ -892,11 +879,11 @@ void OpDispatchBuilder::AVX128_VMOVHP(OpcodeArgs) {
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
///< VMOVHPS/PD mem64, xmm1
|
||||
// Need to store Bits[127:64]. Use a vector element store.
|
||||
auto Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, OpSize::i64Bit, Op->Flags, {.LoadData = false});
|
||||
auto Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
_VStoreVectorElement(OpSize::i128Bit, OpSize::i64Bit, Src1.Low, 1, Dest);
|
||||
} else if (!Op->Src[1].IsGPR()) {
|
||||
///< VMOVHPS/PD xmm2, xmm1, mem64
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], OpSize::i64Bit, Op->Flags, {.LoadData = false});
|
||||
auto Src2 = MakeSegmentAddress(Op, Op->Src[1]);
|
||||
|
||||
// Bits[63:0] come from Src1[63:0]
|
||||
// Bits[127:64] come from Src2[63:0]
|
||||
@@ -1158,7 +1145,7 @@ void OpDispatchBuilder::AVX128_VFCMP(OpcodeArgs) {
|
||||
};
|
||||
|
||||
AVX128_VectorBinaryImpl(Op, GetSrcSize(Op), ElementSize, [this, &Capture](size_t _ElementSize, Ref Src1, Ref Src2) {
|
||||
return VFCMPOpImpl(Capture.Op, _ElementSize, Src1, Src2, Capture.CompType);
|
||||
return VFCMPOpImpl(OpSize::i128Bit, _ElementSize, Src1, Src2, Capture.CompType);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1249,7 +1236,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs) {
|
||||
}
|
||||
|
||||
// If we are storing to memory then we store the size of the element extracted
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
_VStoreVectorElement(OpSize::i128Bit, OverridenElementSize, Src.Low, Index, Dest);
|
||||
}
|
||||
|
||||
@@ -1399,7 +1386,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, size_t ElementSize, const X
|
||||
Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2);
|
||||
} else {
|
||||
// If loading from memory then we only load the element size
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, ElementSize, Op->Flags, {.LoadData = false});
|
||||
auto Src2 = MakeSegmentAddress(Op, Src2Op);
|
||||
Result.Low = _VLoadVectorElement(OpSize::i128Bit, ElementSize, Src1.Low, Index, Src2);
|
||||
}
|
||||
|
||||
@@ -1442,7 +1429,7 @@ void OpDispatchBuilder::AVX128_ShiftDoubleImm(OpcodeArgs, ShiftDirection Dir) {
|
||||
Result = Src;
|
||||
} else if (Shift >= Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
Result.Low = LoadZeroVector(OpSize::i128Bit);
|
||||
Result.High = Result.High;
|
||||
Result.High = Result.Low;
|
||||
} else {
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i128Bit);
|
||||
RefPair Zero {ZeroVector, ZeroVector};
|
||||
@@ -1678,27 +1665,27 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
}
|
||||
}();
|
||||
|
||||
auto Convert = [this](size_t Size, Ref Src, IROps Op) -> Ref {
|
||||
auto Convert = [this](Ref Src, IROps Op) -> Ref {
|
||||
size_t ElementSize = SrcElementSize;
|
||||
if (Widen) {
|
||||
DeriveOp(Extended, Op, _VSXTL(Size, ElementSize, Src));
|
||||
DeriveOp(Extended, Op, _VSXTL(OpSize::i128Bit, ElementSize, Src));
|
||||
Src = Extended;
|
||||
ElementSize <<= 1;
|
||||
}
|
||||
|
||||
return _Vector_SToF(Size, ElementSize, Src);
|
||||
return _Vector_SToF(OpSize::i128Bit, ElementSize, Src);
|
||||
};
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = Convert(Size, Src.Low, IROps::OP_VSXTL);
|
||||
Result.Low = Convert(Src.Low, IROps::OP_VSXTL);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
} else {
|
||||
if (Widen) {
|
||||
Result.High = Convert(Size, Src.Low, IROps::OP_VSXTL2);
|
||||
Result.High = Convert(Src.Low, IROps::OP_VSXTL2);
|
||||
} else {
|
||||
Result.High = Convert(Size, Src.High, IROps::OP_VSXTL);
|
||||
Result.High = Convert(Src.High, IROps::OP_VSXTL);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2173,18 +2160,20 @@ void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref Upper = AVX128_LoadXMMRegister(i, true);
|
||||
_StoreMem(FPRClass, 16, Upper, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
RefPair Pair = LoadContextPair(16, AVXHigh0Index + i);
|
||||
_StoreMemPair(FPRClass, 16, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref YMMHReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
AVX128_StoreXMMRegister(i, YMMHReg, true);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 576);
|
||||
|
||||
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
|
||||
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2234,7 +2223,7 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
|
||||
// For 256-bit, we need to split up the operation. This is nontrivial.
|
||||
// Let's go the simple route here.
|
||||
Ref ZF, CF;
|
||||
Ref ZF, CFInv;
|
||||
Ref ZeroConst = _Constant(0);
|
||||
Ref OneConst = _Constant(1);
|
||||
|
||||
@@ -2278,13 +2267,12 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
|
||||
// ExtGPR will either be [0, 8] or [0, 16] If 0 then set Flag.
|
||||
auto ExtGPR = _VExtractToGPR(OpSize::i128Bit, ElementSize, AddWide, 0);
|
||||
CF = _Select(IR::COND_EQ, ExtGPR, ZeroConst, OneConst, ZeroConst);
|
||||
CFInv = _Select(IR::COND_NEQ, ExtGPR, ZeroConst, OneConst, ZeroConst);
|
||||
}
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
SetNZ_ZeroCV(32, ZF);
|
||||
SetRFLAG(CF, FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
SetCFInverted(CFInv);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -2324,15 +2312,14 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_EQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
// the results of a 16-bit value from the UMaxV, so the 32-bit sign bit is
|
||||
// cleared even if the 16-bit scalars were negative.
|
||||
SetNZ_ZeroCV(32, Test1);
|
||||
SetRFLAG(Test2, FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
SetCFInverted(Test2);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -2488,14 +2475,14 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
|
||||
|
||||
const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit).Low;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit).Low;
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit).Low;
|
||||
|
||||
RefPair Sources[3] = {Dest, Src1, Src2};
|
||||
Ref Sources[3] = {Dest, Src1, Src2};
|
||||
|
||||
DeriveOp(Result_Low, IROp,
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].Low, Sources[Src2Idx - 1].Low, Sources[AddendIdx - 1].Low));
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result_Low));
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable[] = {
|
||||
// Instructions
|
||||
{0x00, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0>},
|
||||
|
||||
{0x08, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0>},
|
||||
|
||||
{0x10, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0>},
|
||||
|
||||
{0x18, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0>},
|
||||
|
||||
{0x20, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0>},
|
||||
|
||||
{0x28, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0>},
|
||||
|
||||
{0x30, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0>},
|
||||
|
||||
{0x38, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0>},
|
||||
{0x50, 8, &OpDispatchBuilder::PUSHREGOp},
|
||||
{0x58, 8, &OpDispatchBuilder::POPOp},
|
||||
{0x68, 1, &OpDispatchBuilder::PUSHOp},
|
||||
{0x69, 1, &OpDispatchBuilder::IMUL2SrcOp},
|
||||
{0x6A, 1, &OpDispatchBuilder::PUSHOp},
|
||||
{0x6B, 1, &OpDispatchBuilder::IMUL2SrcOp},
|
||||
{0x6C, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{0x70, 16, &OpDispatchBuilder::CondJUMPOp},
|
||||
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0x86, 2, &OpDispatchBuilder::XCHGOp},
|
||||
{0x88, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
|
||||
{0x8C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, false>},
|
||||
{0x8D, 1, &OpDispatchBuilder::LEAOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, true>},
|
||||
{0x8F, 1, &OpDispatchBuilder::POPOp},
|
||||
{0x90, 8, &OpDispatchBuilder::XCHGOp},
|
||||
|
||||
{0x98, 1, &OpDispatchBuilder::CDQOp},
|
||||
{0x99, 1, &OpDispatchBuilder::CQOOp},
|
||||
{0x9B, 1, &OpDispatchBuilder::NOPOp},
|
||||
{0x9C, 1, &OpDispatchBuilder::PUSHFOp},
|
||||
{0x9D, 1, &OpDispatchBuilder::POPFOp},
|
||||
{0x9E, 1, &OpDispatchBuilder::SAHFOp},
|
||||
{0x9F, 1, &OpDispatchBuilder::LAHFOp},
|
||||
{0xA4, 2, &OpDispatchBuilder::MOVSOp},
|
||||
|
||||
{0xA6, 2, &OpDispatchBuilder::CMPSOp},
|
||||
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
{0xC2, 2, &OpDispatchBuilder::RETOp},
|
||||
{0xC8, 1, &OpDispatchBuilder::EnterOp},
|
||||
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
|
||||
{0xCC, 2, &OpDispatchBuilder::INTOp},
|
||||
{0xCF, 1, &OpDispatchBuilder::IRETOp},
|
||||
{0xD7, 2, &OpDispatchBuilder::XLATOp},
|
||||
{0xE0, 3, &OpDispatchBuilder::LoopOp},
|
||||
{0xE3, 1, &OpDispatchBuilder::CondJUMPRCXOp},
|
||||
{0xE4, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::CALLOp},
|
||||
{0xE9, 1, &OpDispatchBuilder::JUMPOp},
|
||||
{0xEB, 1, &OpDispatchBuilder::JUMPOp},
|
||||
{0xEC, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xF1, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xF4, 1, &OpDispatchBuilder::INTOp},
|
||||
|
||||
{0xF5, 1, &OpDispatchBuilder::FLAGControlOp},
|
||||
{0xF8, 2, &OpDispatchBuilder::FLAGControlOp},
|
||||
{0xFA, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xFC, 2, &OpDispatchBuilder::FLAGControlOp},
|
||||
};
|
||||
|
||||
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable_64[] = {
|
||||
{0x63, 1, &OpDispatchBuilder::MOVSXDOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
};
|
||||
|
||||
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable_32[] = {
|
||||
{0x06, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x07, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x0E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX>},
|
||||
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x17, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x1E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x1F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x27, 1, &OpDispatchBuilder::DAAOp},
|
||||
{0x2F, 1, &OpDispatchBuilder::DASOp},
|
||||
{0x37, 1, &OpDispatchBuilder::AAAOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::AASOp},
|
||||
{0x40, 8, &OpDispatchBuilder::INCOp},
|
||||
{0x48, 8, &OpDispatchBuilder::DECOp},
|
||||
|
||||
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
|
||||
{0x61, 1, &OpDispatchBuilder::POPAOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
{0xCE, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xD4, 1, &OpDispatchBuilder::AAMOp},
|
||||
{0xD5, 1, &OpDispatchBuilder::AADOp},
|
||||
{0xD6, 1, &OpDispatchBuilder::SALCOp},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,45 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_DDDTable[] = {
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, 4>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, 4>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, 4>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, 4>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, 4>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, 4>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, 4>},
|
||||
|
||||
{0xA0, 1, &OpDispatchBuilder::VPFCMPOp<2>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, 4>},
|
||||
// Can be treated as a move
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, 4>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, 4>},
|
||||
|
||||
{0xB0, 1, &OpDispatchBuilder::VPFCMPOp<0>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, 4>},
|
||||
// Can be treated as a move
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB7, 1, &OpDispatchBuilder::PMULHRWOp},
|
||||
|
||||
{0xBB, 1, &OpDispatchBuilder::PSWAPDOp},
|
||||
{0xBF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 1>},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -6,7 +6,6 @@ desc: Handles x86/64 flag generation
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
@@ -46,6 +45,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
// PF and CF are both stored inverted, so hoist the invert.
|
||||
auto SrcInverted = _Not(OpSize::i32Bit, Src);
|
||||
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
|
||||
@@ -59,31 +61,28 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
// So we write out the whole flags byte to AF without an extract.
|
||||
static_assert(FEXCore::X86State::RFLAG_AF_RAW_LOC == 4);
|
||||
SetRFLAG(Src, FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// PF is stored parity flipped
|
||||
Ref Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC || FlagOffset == FEXCore::X86State::RFLAG_CF_RAW_LOC) {
|
||||
// PF and CF are both stored parity flipped.
|
||||
SetRFLAG(SrcInverted, FlagOffset, FlagOffset, true);
|
||||
} else {
|
||||
SetRFLAG(Src, FlagOffset, FlagOffset, true);
|
||||
}
|
||||
}
|
||||
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Original = _Constant(0);
|
||||
|
||||
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_RAW_LOC)) && (FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
|
||||
// Handle CF first, since it's at bit 0 and hence doesn't need shift or OR.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_RAW_LOC == 0);
|
||||
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_RAW_LOC), "CF always handled");
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_RAW_LOC == 0);
|
||||
Ref Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
for (size_t i = 0; i < FlagOffsets.size(); ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
@@ -114,7 +113,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// instead.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
|
||||
// Set every bit except the bottommost.
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull));
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false, false), _InlineConstant(~1ull));
|
||||
|
||||
// Rotate the bottom bit to the appropriate location for PF, so we get
|
||||
// something like 111P1111. Then invert that to get 000p0000. Then OR that
|
||||
@@ -127,13 +126,13 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
if (GetNZ) {
|
||||
static_assert(FEXCore::X86State::RFLAG_SF_RAW_LOC == (FEXCore::X86State::RFLAG_ZF_RAW_LOC + 1));
|
||||
auto NZCV = GetNZCV();
|
||||
auto NZ = _And(OpSize::i64Bit, NZCV, _Constant(0b11u << 30));
|
||||
auto NZ = _And(OpSize::i64Bit, NZCV, _InlineConstant(0b11u << 30));
|
||||
Original = _Orlshr(OpSize::i64Bit, Original, NZ, 31 - FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
|
||||
// The constant is OR'ed in at the end, to avoid a pointless or xzr, #2.
|
||||
if ((1U << X86State::RFLAG_RESERVED_LOC) & FlagsMask) {
|
||||
Original = _Or(OpSize::i64Bit, Original, _Constant(2));
|
||||
Original = _Or(OpSize::i64Bit, Original, _InlineConstant(2));
|
||||
}
|
||||
|
||||
return Original;
|
||||
@@ -175,22 +174,12 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadPFRaw(bool Invert) {
|
||||
// Read the stored byte. This is the original result (up to 64-bits), it needs
|
||||
// parity calculated.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
Ref OpDispatchBuilder::LoadPFRaw(bool Mask, bool Invert) {
|
||||
// Most blocks do not read parity, so PF optimization is gated on this flag.
|
||||
CurrentHeader->ReadsParity = true;
|
||||
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4);
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2);
|
||||
|
||||
if (Invert) {
|
||||
Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
|
||||
} else {
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
|
||||
}
|
||||
|
||||
return Result;
|
||||
// Evaluate parity on the deferred raw value.
|
||||
return _Parity(GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC), Mask, Invert);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadAF() {
|
||||
@@ -267,34 +256,42 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
|
||||
NZCVDirty = false;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
// If CF not inverted, we use .cc since the increment happens when the
|
||||
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref Res;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
RectifyCarryInvert(false);
|
||||
HandleNZCV_RMW();
|
||||
Res = _AdcWithFlags(OpSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
} else {
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
Ref Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
Ref Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_ULT, Res, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
SetCFInverted(SelectCFInv);
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, false);
|
||||
}
|
||||
|
||||
@@ -303,36 +300,34 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
// Rectify input carry
|
||||
CarryInvert();
|
||||
|
||||
// Arm's subtraction has inverted CF from x86, so rectify the input and
|
||||
// invert the output.
|
||||
RectifyCarryInvert(true);
|
||||
HandleNZCV_RMW();
|
||||
Res = _SbbWithFlags(OpSize, Src1, Src2);
|
||||
|
||||
// Rectify output carry
|
||||
CarryInvert();
|
||||
CFInverted = true;
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, SrcSize * 8, 0, Src1);
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
|
||||
auto Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_ULT, Src1, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
SetCFInverted(SelectCFInv);
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, true);
|
||||
}
|
||||
|
||||
@@ -342,7 +337,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
HandleNZCVWrite();
|
||||
|
||||
@@ -358,12 +353,13 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// If we're updating CF, we need to invert it for correctness. If we're not
|
||||
// updating CF, we need to restore the CF since we stomped over it.
|
||||
// If we're updating CF, we need it to be inverted because SubNZCV is inverted
|
||||
// from x86. If we're not updating CF, we need to restore the CF since we
|
||||
// stomped over it.
|
||||
if (UpdateCF) {
|
||||
CarryInvert();
|
||||
CFInverted = true;
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
SetCFInverted(OldCFInv);
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -371,7 +367,7 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
HandleNZCVWrite();
|
||||
|
||||
@@ -388,8 +384,11 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
CalculatePF(Res);
|
||||
|
||||
// We stomped over CF while calculation flags, restore it.
|
||||
if (!UpdateCF) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
if (UpdateCF) {
|
||||
// Adds match between x86 and arm64.
|
||||
CFInverted = false;
|
||||
} else {
|
||||
SetCFInverted(OldCFInv);
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -404,26 +403,28 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High) {
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
|
||||
// undefined, this does what we need.
|
||||
auto Zero = _Constant(0);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
// undefined, this does what we need after inverting carry.
|
||||
auto Zero = _InlineConstant(0);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
HandleNZCVWrite();
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto Zero = _InlineConstant(0);
|
||||
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
|
||||
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
_SubNZCV(Size, High, Zero);
|
||||
|
||||
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
|
||||
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
@@ -452,7 +453,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
// nothing to do in that case since we already cleared CF above.
|
||||
auto SrcSizeBits = SrcSize * 8;
|
||||
if (Shift < SrcSizeBits) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, SrcSizeBits - Shift, true);
|
||||
SetCFDirect(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -477,11 +478,8 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift - 1, true);
|
||||
}
|
||||
// Extract the last bit shifted in to CF
|
||||
SetCFDirect(Src1, Shift - 1, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
InvalidateAF();
|
||||
@@ -497,11 +495,8 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
// set below.
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift - 1, true);
|
||||
}
|
||||
// Extract the last bit shifted in to CF
|
||||
SetCFDirect(Src1, Shift - 1, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
InvalidateAF();
|
||||
@@ -546,64 +541,6 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(Ref Src) {
|
||||
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
|
||||
// AF are undefined.
|
||||
SetNZ_ZeroCV(GetOpSize(Src), Src);
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, Ref Result) {
|
||||
// CF is cleared if Src is zero, otherwise it's set. However, Src is zero iff
|
||||
// Result is zero, so we can test the result instead. So, CF is just the
|
||||
// inverted ZF.
|
||||
//
|
||||
// ZF/SF/OF set as usual.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto CFOp = GetRFLAG(X86State::RFLAG_ZF_RAW_LOC, true /* Invert */);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF set according to the Src
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
|
||||
|
||||
// The output of BLSMSK is always nonzero, so TST will clear Z (along with C
|
||||
// and O) while setting S.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(Ref Result) {
|
||||
// We need to set ZF while clearing the rest of NZCV. The result of a popcount
|
||||
// is in the range [0, 63]. In particular, it is always positive. So a
|
||||
// combined NZ test will correctly zero SF/CF/OF while setting ZF.
|
||||
SetNZ_ZeroCV(OpSize::i32Bit, Result);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
InvalidatePF_AF();
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
// Test ZF of result, SF is undefined so this is ok.
|
||||
@@ -613,16 +550,7 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// Result is <= SrcSize * 8, we equivalently check if the log2(SrcSize * 8)
|
||||
// bit is set. No masking is needed because no higher bits could be set.
|
||||
unsigned CarryBit = FEXCore::ilog2(SrcSize * 8u);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Result, CarryBit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(Ref Src) {
|
||||
// OF, SF, ZF, AF, PF all zero
|
||||
ZeroNZCV();
|
||||
ZeroPF_AF();
|
||||
|
||||
// CF is set to the incoming source
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
SetCFDirect(Result, CarryBit);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,82 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_66, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_NONE, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 2>},
|
||||
{OPD(PF_38_66, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 2>},
|
||||
{OPD(PF_38_NONE, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 4>},
|
||||
{OPD(PF_38_66, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 4>},
|
||||
{OPD(PF_38_NONE, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_66, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_NONE, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_66, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::PHSUB<2>},
|
||||
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::PHSUB<2>},
|
||||
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::PHSUB<4>},
|
||||
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::PHSUB<4>},
|
||||
{OPD(PF_38_NONE, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_66, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::PSIGN<1>},
|
||||
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::PSIGN<1>},
|
||||
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::PSIGN<2>},
|
||||
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::PSIGN<2>},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::PSIGN<4>},
|
||||
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::PSIGN<4>},
|
||||
{OPD(PF_38_NONE, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x10), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, 1>},
|
||||
{OPD(PF_38_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, 4>},
|
||||
{OPD(PF_38_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, 8>},
|
||||
{OPD(PF_38_66, 0x17), 1, &OpDispatchBuilder::PTestOp},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 1>},
|
||||
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 1>},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 2>},
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 2>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 4>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 4>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<1, 8, true>},
|
||||
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<2, 4, true>},
|
||||
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<2, 8, true>},
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<4, 8, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<4, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 8>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<4>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, false>},
|
||||
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<1, 8, false>},
|
||||
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<2, 4, false>},
|
||||
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<2, 8, false>},
|
||||
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<4, 8, false>},
|
||||
{OPD(PF_38_66, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 8>},
|
||||
{OPD(PF_38_66, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 1>},
|
||||
{OPD(PF_38_66, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 4>},
|
||||
{OPD(PF_38_66, 0x3A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 2>},
|
||||
{OPD(PF_38_66, 0x3B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 4>},
|
||||
{OPD(PF_38_66, 0x3C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 1>},
|
||||
{OPD(PF_38_66, 0x3D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 4>},
|
||||
{OPD(PF_38_66, 0x3E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 2>},
|
||||
{OPD(PF_38_66, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 4>},
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, 4>},
|
||||
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
{OPD(PF_38_66, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
{OPD(PF_38_F3, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,51 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<4>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<8>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<4>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<8>},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<4>},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<8>},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<2>},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 1>},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 2>},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 4>},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 4>},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<1>},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<4>},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<4>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 8>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<8>},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
|
||||
#undef OPD
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,129 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
using X86Tables::OpToIndex;
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_PrimaryGroupTables[] = {
|
||||
// GROUP 1
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>}, // CMP
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
|
||||
// GROUP 2
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, false>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, false>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 2), 1, &OpDispatchBuilder::RCLOp1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 3), 1, &OpDispatchBuilder::RCROp8x1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, true>}, // 1Bit SHR
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, true>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 2), 1, &OpDispatchBuilder::RCLOp1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 3), 1, &OpDispatchBuilder::RCROp1Bit},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, true>}, // 1Bit SHR
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, true>}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, true, true>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::RCLSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::RCRSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
|
||||
|
||||
// GROUP 3
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, &OpDispatchBuilder::MULOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 5), 1, &OpDispatchBuilder::IMULOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 4), 1, &OpDispatchBuilder::MULOp}, // MUL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 5), 1, &OpDispatchBuilder::IMULOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
|
||||
|
||||
// GROUP 4
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_4, OpToIndex(0xFE), 0), 1, &OpDispatchBuilder::INCOp}, // INC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_4, OpToIndex(0xFE), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
|
||||
|
||||
// GROUP 5
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, &OpDispatchBuilder::INCOp}, // INC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, &OpDispatchBuilder::CALLAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, &OpDispatchBuilder::JUMPAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp},
|
||||
|
||||
// GROUP 11
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 1>},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,161 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
constexpr uint16_t PF_NONE = 0;
|
||||
constexpr uint16_t PF_F3 = 1;
|
||||
constexpr uint16_t PF_66 = 2;
|
||||
constexpr uint16_t PF_F2 = 3;
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryGroupTables[] = {
|
||||
// GROUP 6
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_F2, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 4), 1, &OpDispatchBuilder::SMSWOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTSet>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTClear>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_66, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F2, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTComplement>},
|
||||
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 7), 1, &OpDispatchBuilder::RDPIDOp},
|
||||
|
||||
// GROUP 12
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 2>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 2>},
|
||||
|
||||
// GROUP 13
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 4>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 4>},
|
||||
|
||||
// GROUP 14
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 8>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 3), 1, &OpDispatchBuilder::PSRLDQ},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLLDQ},
|
||||
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 0), 1, &OpDispatchBuilder::FXSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 1), 1, &OpDispatchBuilder::FXRStoreOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 2), 1, &OpDispatchBuilder::LDMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 3), 1, &OpDispatchBuilder::STMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 4), 1, &OpDispatchBuilder::XSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::LoadFenceOrXRSTOR}, // LFENCE (or XRSTOR)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, // MFENCE (or XSAVEOPT)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, // SFENCE (or CLFLUSH)
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 6), 1, &OpDispatchBuilder::CLWB},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 7), 1, &OpDispatchBuilder::CLFLUSHOPT},
|
||||
|
||||
// GROUP 16
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F3, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_66, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
// GROUP P
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 3), 5, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F3, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_66, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryGroupTables_64[] = {
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,22 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryModRMTables[] = {
|
||||
// REG /1
|
||||
{((0 << 3) | 0), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{((0 << 3) | 1), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
// REG /2
|
||||
{((1 << 3) | 0), 1, &OpDispatchBuilder::XGetBVOp},
|
||||
|
||||
// REG /3
|
||||
{((2 << 3) | 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 0), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,329 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x0B, 1, &OpDispatchBuilder::INTOp},
|
||||
{0x0E, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp}, // NOP with ModRM
|
||||
|
||||
{0x20, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{0x30, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x31, 1, &OpDispatchBuilder::RDTSCOp},
|
||||
{0x32, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x34, 3, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
{0x80, 16, &OpDispatchBuilder::CondJUMPOp},
|
||||
{0x90, 16, &OpDispatchBuilder::SETccOp},
|
||||
{0xA2, 1, &OpDispatchBuilder::CPUIDOp},
|
||||
{0xA3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTNone>}, // BT
|
||||
{0xA4, 1, &OpDispatchBuilder::SHLDImmediateOp},
|
||||
{0xA5, 1, &OpDispatchBuilder::SHLDOp},
|
||||
{0xAB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTSet>}, // BTS
|
||||
{0xAC, 1, &OpDispatchBuilder::SHRDImmediateOp},
|
||||
{0xAD, 1, &OpDispatchBuilder::SHRDOp},
|
||||
{0xAF, 1, &OpDispatchBuilder::IMUL1SrcOp},
|
||||
{0xB0, 2, &OpDispatchBuilder::CMPXCHGOp}, // CMPXCHG
|
||||
{0xB3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTClear>}, // BTR
|
||||
{0xB6, 2, &OpDispatchBuilder::MOVZXOp},
|
||||
{0xBB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 0, BTAction::BTComplement>}, // BTC
|
||||
{0xBC, 1, &OpDispatchBuilder::BSFOp}, // BSF
|
||||
{0xBD, 1, &OpDispatchBuilder::BSROp}, // BSF
|
||||
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
|
||||
{0xC0, 2, &OpDispatchBuilder::XADDOp},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 2>},
|
||||
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
|
||||
|
||||
// SSE
|
||||
{0x10, 2, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x12, 2, &OpDispatchBuilder::MOVLPOp},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 4>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 4>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, 4>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, 4>},
|
||||
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, 4>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 16>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, 8, 4, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, 4>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, 4>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, 4>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 1>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 2>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 4>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<2>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 1>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 2>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 4>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<2>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 1>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 2>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 4>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<4>},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFW8ByteOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 1>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 2>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 4>},
|
||||
{0x77, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, 4>},
|
||||
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 2>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 4>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 8>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 8>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, 2>},
|
||||
{0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 1>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 2>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 1>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 8>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 1>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 2>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 1>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 1>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 2>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 4>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 2>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 1>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 2>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 2>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 8>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 1>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 2>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 2>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 2>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 4>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 8>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<4, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 1>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 2>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 4>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 8>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 1>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 2>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 4>},
|
||||
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSSOp},
|
||||
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
|
||||
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<4>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<8, 4>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, false>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
|
||||
{0xBC, 1, &OpDispatchBuilder::TZCNT},
|
||||
{0xBD, 1, &OpDispatchBuilder::LZCNT},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<4>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, true>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
|
||||
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<8>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>},
|
||||
// x52 = Invalid
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<4, 8>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, 4>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<4>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<4>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<8>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, true>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x12, 2, &OpDispatchBuilder::MOVLPOp},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 8>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 8>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<8>},
|
||||
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, 8>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, 8>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 16>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, 4, 8, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, 8>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, 8>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, 8>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 1>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 2>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 4>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<2>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 1>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 2>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 4>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<2>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 1>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 2>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 4>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<4>},
|
||||
{0x6C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 8>},
|
||||
{0x6D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 8>},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFDOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 1>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 2>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 4>},
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, 8>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<8>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 2>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, 8>},
|
||||
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<8>},
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 2>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 4>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 8>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 8>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, 2>},
|
||||
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 1>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 2>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 1>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 16>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 1>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 2>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 1>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 1>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 2>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 4>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 2>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 1>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 2>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 2>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 16>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 1>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 2>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 2>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 2>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 4>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 8>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<4, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 1>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 2>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 4>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 8>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 1>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 2>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 4>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable_64[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SyscallOp, true>},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable_32[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::NOPOp},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -0,0 +1,26 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define OPD(map_select, pp, opcode) (((map_select - 1) << 10) | (pp << 8) | (opcode))
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_VEXTable[] = {
|
||||
{OPD(2, 0b00, 0xF2), 1, &OpDispatchBuilder::ANDNBMIOp}, {OPD(2, 0b00, 0xF5), 1, &OpDispatchBuilder::BZHI},
|
||||
{OPD(2, 0b10, 0xF5), 1, &OpDispatchBuilder::PEXT}, {OPD(2, 0b11, 0xF5), 1, &OpDispatchBuilder::PDEP},
|
||||
{OPD(2, 0b11, 0xF6), 1, &OpDispatchBuilder::MULX}, {OPD(2, 0b00, 0xF7), 1, &OpDispatchBuilder::BEXTRBMIOp},
|
||||
{OPD(2, 0b01, 0xF7), 1, &OpDispatchBuilder::BMI2Shift}, {OPD(2, 0b10, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
{OPD(2, 0b11, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
|
||||
{OPD(3, 0b11, 0xF0), 1, &OpDispatchBuilder::RORX},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
#define OPD(group, pp, opcode) (((group - X86Tables::InstType::TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
constexpr std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_VEXGroupTable[] = {
|
||||
{OPD(X86Tables::InstType::TYPE_VEX_GROUP_17, 0, 0b001), 1, &OpDispatchBuilder::BLSRBMIOp},
|
||||
{OPD(X86Tables::InstType::TYPE_VEX_GROUP_17, 0, 0b010), 1, &OpDispatchBuilder::BLSMSKBMIOp},
|
||||
{OPD(X86Tables::InstType::TYPE_VEX_GROUP_17, 0, 0b011), 1, &OpDispatchBuilder::BLSIBMIOp},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -106,7 +106,7 @@ void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) {
|
||||
} else {
|
||||
// If the destination is a GPR then the source is memory
|
||||
// xmm1[127:64] = src
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref Src = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags);
|
||||
auto Result = _VLoadVectorElement(16, 8, Dest, 1, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -115,7 +115,7 @@ void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) {
|
||||
// In this case memory is the destination and the high bits of the XMM are source
|
||||
// Mem64 = xmm1[127:64]
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, 8, Op->Flags, {.LoadData = false});
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
_VStoreVectorElement(16, 8, Src, 1, Dest);
|
||||
}
|
||||
}
|
||||
@@ -144,7 +144,7 @@ void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 16, 16);
|
||||
} else {
|
||||
auto DstSize = GetDstSize(Op);
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = 8, .LoadData = false});
|
||||
Ref Src = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags);
|
||||
auto Result = _VLoadVectorElement(16, 8, Dest, 0, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -702,8 +702,7 @@ void OpDispatchBuilder::MOVQMMXOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
uint8_t NumElements = Size / ElementSize;
|
||||
|
||||
@@ -752,9 +751,6 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::MOVMSKOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::MOVMSKOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
@@ -780,8 +776,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -791,13 +786,7 @@ void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PUNPCKLOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKLOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKLOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKLOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -817,13 +806,7 @@ void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
@@ -832,13 +815,7 @@ void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PUNPCKHOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKHOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKHOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKHOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -858,11 +835,6 @@ void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::GeneratePSHUFBMask(uint8_t SrcSize) {
|
||||
// PSHUFB doesn't 100% match VTBL behaviour
|
||||
// VTBL will set the element zero if the index is greater than
|
||||
@@ -949,8 +921,7 @@ void OpDispatchBuilder::PSHUFW8ByteOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template<bool Low>
|
||||
void OpDispatchBuilder::PSHUFWOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSHUFWOp(OpcodeArgs, bool Low) {
|
||||
constexpr auto IdentityCopy = 0b11'10'01'00;
|
||||
|
||||
uint16_t Shuffle = Op->Src[1].Data.Literal.Value;
|
||||
@@ -1000,9 +971,6 @@ void OpDispatchBuilder::PSHUFWOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSHUFWOp<false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSHUFWOp<true>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle) {
|
||||
constexpr auto IdentityCopy = 0b11'10'01'00;
|
||||
|
||||
@@ -1222,8 +1190,7 @@ void OpDispatchBuilder::PSHUFDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Single128Bit4ByteVectorShuffle(Src, Shuffle), -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize, bool Low>
|
||||
void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs, size_t ElementSize, bool Low) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
auto Shuffle = Op->Src[1].Literal();
|
||||
@@ -1271,9 +1238,6 @@ void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::VPSHUFWOp<2, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSHUFWOp<2, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSHUFWOp<4, true>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle) {
|
||||
// Since 256-bit variants and up don't lane cross, we can construct
|
||||
@@ -1466,8 +1430,7 @@ Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize
|
||||
return Dest;
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::SHUFOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHUFOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Src1Node = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src2Node = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
uint8_t Shuffle = Op->Src[1].Literal();
|
||||
@@ -1475,11 +1438,8 @@ void OpDispatchBuilder::SHUFOp(OpcodeArgs) {
|
||||
Ref Result = SHUFOpImpl(Op, GetDstSize(Op), ElementSize, Src1Node, Src2Node, Shuffle);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::SHUFOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::SHUFOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VSHUFOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VSHUFOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Src1Node = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2Node = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
uint8_t Shuffle = Op->Src[2].Literal();
|
||||
@@ -1487,8 +1447,6 @@ void OpDispatchBuilder::VSHUFOp(OpcodeArgs) {
|
||||
Ref Result = SHUFOpImpl(Op, GetDstSize(Op), ElementSize, Src1Node, Src2Node, Shuffle);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::VSHUFOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VSHUFOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VANDNOp(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
@@ -1524,8 +1482,7 @@ template void OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, 4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, 4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, 8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
Ref Result {};
|
||||
|
||||
@@ -1544,12 +1501,6 @@ void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VBROADCASTOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<8>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<16>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, const X86Tables::DecodedOperand& Imm) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
@@ -1564,7 +1515,7 @@ Ref OpDispatchBuilder::PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tabl
|
||||
}
|
||||
|
||||
// If loading from memory then we only load the element size
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, ElementSize, Op->Flags, {.LoadData = false});
|
||||
Ref Src2 = MakeSegmentAddress(Op, Src2Op);
|
||||
return _VLoadVectorElement(Size, ElementSize, Src1, Index, Src2);
|
||||
}
|
||||
|
||||
@@ -1660,8 +1611,7 @@ void OpDispatchBuilder::VINSERTPSOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PExtrOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
@@ -1672,7 +1622,7 @@ void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
// is the same except that REX.W or VEX.W is set to 1. Incredibly frustrating.
|
||||
// Use the destination size as the element size in this case.
|
||||
size_t OverridenElementSize = ElementSize;
|
||||
if constexpr (ElementSize == 4) {
|
||||
if (ElementSize == 4) {
|
||||
OverridenElementSize = DstSize;
|
||||
}
|
||||
|
||||
@@ -1689,15 +1639,10 @@ void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
// If we are storing to memory then we store the size of the element extracted
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
_VStoreVectorElement(16, OverridenElementSize, Src, Index, Dest);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PExtrOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PExtrOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PExtrOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PExtrOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VEXTRACT128Op(OpcodeArgs) {
|
||||
const auto DstIsXMM = Op->Dest.IsGPR();
|
||||
const auto StoreSize = DstIsXMM ? 32 : 16;
|
||||
@@ -1761,8 +1706,7 @@ Ref OpDispatchBuilder::PSRLDOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref
|
||||
return _VUShrSWide(Size, ElementSize, Src, ShiftVec);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRLDOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = PSRLDOpImpl(Op, ElementSize, Dest, Src);
|
||||
@@ -1770,12 +1714,7 @@ void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRLDOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLDOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLDOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRLDOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRLDOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1789,12 +1728,7 @@ void OpDispatchBuilder::VPSRLDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRLDOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLDOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLDOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRLI(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRLI(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
if (ShiftConstant == 0) [[unlikely]] {
|
||||
// Nothing to do, value is already in Dest.
|
||||
@@ -1808,12 +1742,7 @@ void OpDispatchBuilder::PSRLI(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Shift, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRLI<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLI<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLI<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRLIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRLIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
@@ -1832,10 +1761,6 @@ void OpDispatchBuilder::VPSRLIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRLIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLIOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLIOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PSLLIImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Shift) {
|
||||
if (Shift == 0) [[unlikely]] {
|
||||
// If zero-shift then just return the source.
|
||||
@@ -1845,8 +1770,7 @@ Ref OpDispatchBuilder::PSLLIImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64
|
||||
return _VShlI(Size, ElementSize, Src, Shift);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSLLI(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSLLI(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
if (ShiftConstant == 0) [[unlikely]] {
|
||||
// Nothing to do, value is already in Dest.
|
||||
@@ -1859,12 +1783,7 @@ void OpDispatchBuilder::PSLLI(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSLLI<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLLI<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLLI<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSLLIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSLLIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
@@ -1878,10 +1797,6 @@ void OpDispatchBuilder::VPSLLIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSLLIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLIOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLIOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PSLLImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
@@ -1889,8 +1804,7 @@ Ref OpDispatchBuilder::PSLLImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref Shi
|
||||
return _VUShlSWide(Size, ElementSize, Src, ShiftVec);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSLL(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSLL(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = PSLLImpl(Op, ElementSize, Dest, Src);
|
||||
@@ -1898,12 +1812,7 @@ void OpDispatchBuilder::PSLL(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSLL<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLL<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLL<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSLLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSLLOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1917,10 +1826,6 @@ void OpDispatchBuilder::VPSLLOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSLLOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
@@ -1928,8 +1833,7 @@ Ref OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref S
|
||||
return _VSShrSWide(Size, ElementSize, Src, ShiftVec);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRAOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = PSRAOpImpl(Op, ElementSize, Dest, Src);
|
||||
@@ -1937,11 +1841,7 @@ void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRAOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRAOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRAOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRAOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1955,9 +1855,6 @@ void OpDispatchBuilder::VPSRAOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRAOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRAOp<4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PSRLDQ(OpcodeArgs) {
|
||||
const uint64_t Shift = Op->Src[1].Literal();
|
||||
if (Shift == 0) [[unlikely]] {
|
||||
@@ -2059,8 +1956,7 @@ void OpDispatchBuilder::VPSLLDQOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRAIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRAIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t Shift = Op->Src[1].Literal();
|
||||
if (Shift == 0) [[unlikely]] {
|
||||
// Nothing to do, value is already in Dest.
|
||||
@@ -2074,11 +1970,7 @@ void OpDispatchBuilder::PSRAIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRAIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRAIOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRAIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRAIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t Shift = Op->Src[1].Literal();
|
||||
const auto Size = GetDstSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
@@ -2097,9 +1989,6 @@ void OpDispatchBuilder::VPSRAIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRAIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRAIOp<4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::AVXVariableShiftImpl(OpcodeArgs, IROps IROp) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
@@ -2529,9 +2418,7 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) {
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType) {
|
||||
Ref Result {};
|
||||
switch (CompType & 0x7) {
|
||||
case 0x0: // EQ
|
||||
@@ -2567,7 +2454,7 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags);
|
||||
const uint8_t CompType = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
Ref Result = VFCMPOpImpl(Op, ElementSize, Dest, Src, CompType);
|
||||
Ref Result = VFCMPOpImpl(OpSizeFromSrc(Op), ElementSize, Dest, Src, CompType);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -2585,7 +2472,7 @@ void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) {
|
||||
|
||||
Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Ref Result = VFCMPOpImpl(Op, ElementSize, Src1, Src2, CompType);
|
||||
Ref Result = VFCMPOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2, CompType);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -2736,37 +2623,33 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
Ref MMReg = LoadContext(MM0Index + i);
|
||||
|
||||
_StoreMem(FPRClass, 16, MMReg, MemBase, _Constant(i * 16 + 32), 16, MEM_OFFSET_SXTX, 1);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
RefPair MMRegs = LoadContextPair(16, MM0Index + i);
|
||||
_StoreMemPair(FPRClass, 16, MMRegs.Low, MMRegs.High, MemBase, i * 16 + 32);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveSSEState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref XMMReg = LoadXMMRegister(i);
|
||||
|
||||
_StoreMem(FPRClass, 16, XMMReg, MemBase, _Constant(i * 16 + 160), 16, MEM_OFFSET_SXTX, 1);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
_StoreMemPair(FPRClass, 16, LoadXMMRegister(i), LoadXMMRegister(i + 1), MemBase, i * 16 + 160);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(Ref MemBase) {
|
||||
_StoreMem(GPRClass, 4, GetMXCSR(), MemBase, _Constant(24), 4, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// Store the mask for all bits.
|
||||
_StoreMem(GPRClass, 4, _Constant(0xFFFF), MemBase, _Constant(28), 4, MEM_OFFSET_SXTX, 1);
|
||||
// Store MXCSR and the mask for all bits.
|
||||
_StoreMemPair(GPRClass, 4, GetMXCSR(), _Constant(0xFFFF), MemBase, 24);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref Upper = _VDupElement(32, 16, LoadXMMRegister(i), 1);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
Ref Upper0 = _VDupElement(32, 16, LoadXMMRegister(i + 0), 1);
|
||||
Ref Upper1 = _VDupElement(32, 16, LoadXMMRegister(i + 1), 1);
|
||||
|
||||
_StoreMem(FPRClass, 16, Upper, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemPair(FPRClass, 16, Upper0, Upper1, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2868,18 +2751,22 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
StoreContext(AbridgedFTWIndex, _LoadMem(GPRClass, 1, MemBase, _Constant(4), 2, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 32), 16, MEM_OFFSET_SXTX, 1);
|
||||
StoreContext(MM0Index + i, MMReg);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
auto MMRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 32);
|
||||
|
||||
StoreContext(MM0Index + i, MMRegs.Low);
|
||||
StoreContext(MM0Index + i + 1, MMRegs.High);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreSSEState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref XMMReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 160), 16, MEM_OFFSET_SXTX, 1);
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto XMMRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 160);
|
||||
|
||||
StoreXMMRegister(i, XMMRegs.Low);
|
||||
StoreXMMRegister(i + 1, XMMRegs.High);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2896,11 +2783,12 @@ void OpDispatchBuilder::RestoreMXCSRState(Ref MXCSR) {
|
||||
void OpDispatchBuilder::RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref XMMReg = LoadXMMRegister(i);
|
||||
Ref YMMHReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
Ref YMM = _VInsElement(32, 16, 1, 0, XMMReg, YMMHReg);
|
||||
StoreXMMRegister(i, YMM);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
Ref XMMReg0 = LoadXMMRegister(i + 0);
|
||||
Ref XMMReg1 = LoadXMMRegister(i + 1);
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 576);
|
||||
StoreXMMRegister(i + 0, _VInsElement(32, 16, 1, 0, XMMReg0, YMMHRegs.Low));
|
||||
StoreXMMRegister(i + 1, _VInsElement(32, 16, 1, 0, XMMReg1, YMMHRegs.High));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3015,8 +2903,7 @@ void OpDispatchBuilder::PACKUSOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::PACKUSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PACKUSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPACKUSOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPACKUSOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3032,9 +2919,6 @@ void OpDispatchBuilder::VPACKUSOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPACKUSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPACKUSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PACKSSOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -3047,8 +2931,7 @@ void OpDispatchBuilder::PACKSSOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::PACKSSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPACKSSOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPACKSSOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3064,9 +2947,6 @@ void OpDispatchBuilder::VPACKSSOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPACKSSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PMULLOpImpl(OpSize Size, size_t ElementSize, bool Signed, Ref Src1, Ref Src2) {
|
||||
if (Size == OpSize::i64Bit) {
|
||||
if (Signed) {
|
||||
@@ -3505,8 +3385,7 @@ void OpDispatchBuilder::HSUBP(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::HSUBP<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::HSUBP<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VHSUBPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VHSUBPOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3523,9 +3402,6 @@ void OpDispatchBuilder::VHSUBPOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VHSUBPOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VHSUBPOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PHSUBOpImpl(OpSize Size, Ref Src1, Ref Src2, size_t ElementSize) {
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
@@ -3543,8 +3419,7 @@ void OpDispatchBuilder::PHSUB(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::PHSUB<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PHSUB<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPHSUBOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPHSUBOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3558,9 +3433,6 @@ void OpDispatchBuilder::VPHSUBOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPHSUBOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPHSUBOp<4>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2) {
|
||||
const uint8_t ElementSize = 2;
|
||||
|
||||
@@ -4027,8 +3899,7 @@ template void OpDispatchBuilder::VectorBlend<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorBlend<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorBlend<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -4047,14 +3918,10 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::VectorVariableBlend<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorVariableBlend<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorVariableBlend<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, size_t ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
constexpr auto ElementSizeBits = ElementSize * 8;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
@@ -4067,9 +3934,6 @@ void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs) {
|
||||
Ref Result = _VBSL(SrcSize, Shifted, Src2, Src1);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::AVXVectorVariableBlend<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVectorVariableBlend<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVectorVariableBlend<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
// Invalidate deferred flags early
|
||||
@@ -4088,15 +3952,14 @@ void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_EQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
// the results of a 16-bit value from the UMaxV, so the 32-bit sign bit is
|
||||
// cleared even if the 16-bit scalars were negative.
|
||||
SetNZ_ZeroCV(32, Test1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Test2);
|
||||
|
||||
SetCFInverted(Test2);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -4130,12 +3993,11 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, size_t ElementSize, Ref Src1
|
||||
Ref ZeroConst = _Constant(0);
|
||||
Ref OneConst = _Constant(1);
|
||||
|
||||
Ref CFResult = _Select(IR::COND_EQ, AndNotGPR, ZeroConst, OneConst, ZeroConst);
|
||||
Ref CFInv = _Select(IR::COND_NEQ, AndNotGPR, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
SetNZ_ZeroCV(32, AndGPR);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFResult);
|
||||
|
||||
SetCFInverted(CFInv);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -4890,8 +4752,7 @@ void OpDispatchBuilder::VZEROOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Selector = Op->Src[1].Literal() & 0xFF;
|
||||
@@ -4899,7 +4760,7 @@ void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = LoadZeroVector(DstSize);
|
||||
|
||||
if constexpr (ElementSize == 8) {
|
||||
if (ElementSize == 8) {
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, Selector & 0b0001, Result, Src);
|
||||
Result = _VInsElement(DstSize, ElementSize, 1, (Selector & 0b0010) >> 1, Result, Src);
|
||||
|
||||
@@ -4924,9 +4785,6 @@ void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPERMILImmOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPERMILImmOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, size_t ElementSize, Ref Src, Ref Indices) {
|
||||
// NOTE: See implementation of VPERMD for the gist of what we do to make this work.
|
||||
//
|
||||
@@ -5068,6 +4926,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
|
||||
// Set all of the necessary flags. NZCV stored in bits 28...31 like the hw op.
|
||||
SetNZCV(IntermediateResult);
|
||||
CFInverted = false;
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -150,8 +150,6 @@ void OpDispatchBuilder::FSTToStack(OpcodeArgs) {
|
||||
// Store integer to memory (possibly with truncation)
|
||||
void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
// FIXME(pmatos): is there any advantage of using STORESTACKMEMORY here?
|
||||
// Do we need STORESTACKMEMORY at all?
|
||||
Ref Data = _ReadStackValue(0);
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
|
||||
@@ -177,16 +175,15 @@ void OpDispatchBuilder::FADD(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -208,16 +205,15 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -246,16 +242,15 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -288,16 +283,15 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
}
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -622,7 +616,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatch
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetCFDirect(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
|
||||
@@ -22,26 +22,6 @@ class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
// Functions in X87.cpp (no change required)
|
||||
// GetX87Top
|
||||
// SetX87ValidTag
|
||||
// GetX87ValidTag
|
||||
// GetX87Tag (will need changing once special tag handling is implemented)
|
||||
// SetX87FTW
|
||||
// GetX87FTW (will need changing once special tag handling is implemented)
|
||||
// SetX87Top
|
||||
// X87ModifySTP
|
||||
// EMMS
|
||||
// FFREE
|
||||
// FNSTENV
|
||||
// FSTCW
|
||||
// LDSW
|
||||
// FNSTSW
|
||||
// FXCH
|
||||
// FCMOV
|
||||
// FST(register to register)
|
||||
// FCHS
|
||||
|
||||
void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
// Init host rounding mode to zero
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
@@ -31,32 +31,9 @@ X86GeneratedCode::X86GeneratedCode() {
|
||||
0x0F, 0x37, // CALLBACKRET FEX Instruction
|
||||
};
|
||||
|
||||
// Signal return handlers need to be bit-exact to what the Linux kernel provides in VDSO.
|
||||
// GDB and unwinding libraries key off of these instructions to understand if the stack frame is a signal frame or not.
|
||||
// This two code sections match exactly what libSegFault expects.
|
||||
//
|
||||
// Typically this handlers are provided by the 32-bit VDSO thunk library, but that isn't available in all cases.
|
||||
// Falling back to this generated code segment still allows a backtrace to work, just might not show
|
||||
// the symbol as VDSO since there is no ELF to parse.
|
||||
constexpr std::array<uint8_t, 9> sigreturn_32_code = {
|
||||
0x58, // pop eax
|
||||
0xb8, 0x77, 0x00, 0x00, 0x00, // mov eax, 0x77
|
||||
0xcd, 0x80, // int 0x80
|
||||
0x90, // nop
|
||||
};
|
||||
|
||||
constexpr std::array<uint8_t, 7> rt_sigreturn_32_code = {
|
||||
0xb8, 0xad, 0x00, 0x00, 0x00, // mov eax, 0xad
|
||||
0xcd, 0x80, // int 0x80
|
||||
};
|
||||
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
sigreturn_32 = CallbackReturn + SignalReturnCode.size();
|
||||
rt_sigreturn_32 = sigreturn_32 + sigreturn_32_code.size();
|
||||
|
||||
memcpy(reinterpret_cast<void*>(CallbackReturn), &SignalReturnCode.at(0), SignalReturnCode.size());
|
||||
memcpy(reinterpret_cast<void*>(sigreturn_32), &sigreturn_32_code.at(0), sigreturn_32_code.size());
|
||||
memcpy(reinterpret_cast<void*>(rt_sigreturn_32), &rt_sigreturn_32_code.at(0), rt_sigreturn_32_code.size());
|
||||
|
||||
mprotect(CodePtr, CODE_SIZE, PROT_READ);
|
||||
#endif
|
||||
|
||||
@@ -17,8 +17,6 @@ public:
|
||||
~X86GeneratedCode();
|
||||
|
||||
uint64_t CallbackReturn {};
|
||||
uint64_t sigreturn_32 {};
|
||||
uint64_t rt_sigreturn_32 {};
|
||||
|
||||
private:
|
||||
void* CodePtr {};
|
||||
|
||||
@@ -14,12 +14,14 @@ namespace FEXCore::X86Tables {
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializePrimaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode);
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
InitializeSecondaryGroupTables(Mode);
|
||||
InitializePrimaryGroupTables(Mode);
|
||||
InitializeH0F3ATables(Mode);
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/BaseTables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -236,6 +237,7 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), BaseOpTable, std::size(BaseOpTable));
|
||||
IR::InstallToTable(Table, IR::OpDispatch_BaseOpTable);
|
||||
|
||||
return Table;
|
||||
}();
|
||||
@@ -301,9 +303,11 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_64, std::size(BaseOpTable_64));
|
||||
IR::InstallToTable(BaseOps, IR::OpDispatch_BaseOpTable_64);
|
||||
}
|
||||
else {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_32, std::size(BaseOpTable_32));
|
||||
IR::InstallToTable(BaseOps, IR::OpDispatch_BaseOpTable_32);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/DDDTables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -54,6 +55,8 @@ std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_DDDTable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -1,37 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: frontend|x86-tables
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct EVEXTable[] = {
|
||||
{0x10, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x11, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"VBROADCASTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x19, 1, X86InstInfo{"VBROADCASTD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x1A, 1, X86InstInfo{"VBROADCASTSD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x1B, 1, X86InstInfo{"VBROADCASTF64X4", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x28, 1, X86InstInfo{"VMOVAPS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x29, 1, X86InstInfo{"VMOVAPS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"VBROADCASTQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"VMOVDQU64", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x73, 1, X86InstInfo{"VPSLLDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x7F, 1, X86InstInfo{"VMOVDQU64", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0xE7, 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), EVEXTable, std::size(EVEXTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/H0F38Tables.h"
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
@@ -119,6 +120,8 @@ std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), H0F38Table, std::size(H0F38Table));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F38Table);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/H0F3ATables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -55,6 +56,8 @@ std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -69,6 +72,7 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable_64, std::size(H0F3ATable_64));
|
||||
IR::InstallToTable(H0F3ATableOps, IR::OpDispatch_H0F3ATable_64);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/PrimaryGroupTables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -144,6 +145,8 @@ std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps = []() co
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_PrimaryGroupTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/SecondaryGroupTables.h"
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
@@ -488,7 +489,15 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryGroupTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeSecondaryGroupTables(Context::OperatingMode Mode) {
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
IR::InstallToTable(SecondInstGroupOps, IR::OpDispatch_SecondaryGroupTables_64);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/SecondaryModRMTables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -56,6 +57,8 @@ std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []()
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryModRMTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/SecondaryTables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
@@ -270,6 +271,8 @@ auto BaseOpsLambda = []() consteval {
|
||||
|
||||
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_TwoByteOpTable);
|
||||
|
||||
return Table;
|
||||
};
|
||||
|
||||
@@ -297,7 +300,7 @@ std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
|
||||
{0x2E, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x30, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -359,6 +362,7 @@ std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
|
||||
|
||||
GenerateTableWithCopy(&Table.at(0), RepModOpTable, std::size(RepModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryRepModTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -383,7 +387,7 @@ std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
|
||||
{0x2E, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x30, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -440,6 +444,7 @@ std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
|
||||
|
||||
GenerateTableWithCopy(&Table.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryRepNEModTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -595,6 +600,7 @@ std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps = []() consteval
|
||||
|
||||
GenerateTableWithCopy(&Table.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_SecondaryOpSizeModTables);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -620,12 +626,16 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
|
||||
IR::InstallToTable(SecondBaseOps, IR::OpDispatch_TwoByteOpTable_64);
|
||||
}
|
||||
else {
|
||||
LateInitCopyTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
|
||||
IR::InstallToTable(SecondBaseOps, IR::OpDispatch_TwoByteOpTable_32);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/OpcodeDispatcher/VEXTables.h"
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -489,6 +490,8 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), VEXTable, std::size(VEXTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_VEXTable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
@@ -521,6 +524,7 @@ std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps = []() conste
|
||||
|
||||
GenerateTable(&Table.at(0), VEXGroupTable, std::size(VEXGroupTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_VEXGroupTable);
|
||||
return Table;
|
||||
}();
|
||||
}
|
||||
@@ -474,8 +474,6 @@ constexpr size_t MAX_XOP_TABLE_SIZE = (1 << 13);
|
||||
// group select (2 bits for now) | modrm opcode (3 bits)
|
||||
constexpr size_t MAX_XOP_GROUP_TABLE_SIZE = (1 << 6);
|
||||
|
||||
constexpr size_t MAX_EVEX_TABLE_SIZE = 256;
|
||||
|
||||
extern std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps;
|
||||
extern std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps;
|
||||
extern std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps;
|
||||
@@ -498,9 +496,6 @@ extern std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps;
|
||||
extern std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps;
|
||||
extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
|
||||
// EVEX
|
||||
extern std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
|
||||
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
@@ -518,7 +513,10 @@ constexpr static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInf
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_MSG_A_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if (FinalTable[OpNum + i].OpcodeDispatcher) {
|
||||
LOGMAN_MSG_A_FMT("Already installed an OpcodeDispatcher for 0x{:x}", OpNum + i);
|
||||
}
|
||||
FinalTable[OpNum + i] = Info;
|
||||
}
|
||||
@@ -533,7 +531,7 @@ constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86T
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_MSG_A_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if (Info.Type == TYPE_COPY_OTHER) {
|
||||
FinalTable[OpNum + i] = OtherLocal[OpNum + i];
|
||||
@@ -568,7 +566,7 @@ constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86Tables
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_MSG_A_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if ((OpNum & 0b11'000'000) == 0b11'000'000) {
|
||||
// If the mod field is 0b11 then it is a regular op
|
||||
|
||||
@@ -17,9 +17,34 @@
|
||||
#include <shared_mutex>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugData;
|
||||
}
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
};
|
||||
|
||||
struct DebugDataGuestOpcode {
|
||||
uint64_t GuestEntryOffset;
|
||||
ptrdiff_t HostEntryOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Contains debug data for a block of code for later debugger analysis
|
||||
*
|
||||
* Needs to remain around for as long as the code could be executed at least
|
||||
*/
|
||||
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
fextl::vector<DebugDataSubblock> Subblocks;
|
||||
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
|
||||
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
|
||||
};
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
@@ -44,7 +44,7 @@
|
||||
" * Textual class to group IR ops by type",
|
||||
"* DestClass",
|
||||
" * SSA class of the return when the return type is `SSA`",
|
||||
" * Not used if the destination type is one of {GPR, GPRPair, FPR}",
|
||||
" * Not used if the destination type is one of {GPR, FPR}",
|
||||
"* DestSize",
|
||||
" * The size of the destination type",
|
||||
"* EmitValidation",
|
||||
@@ -67,6 +67,8 @@
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_TSTZ = 14 /* bit test zero */",
|
||||
"constexpr uint8_t COND_TSTNZ = 15 /* bit test nonzero */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
@@ -81,7 +83,6 @@
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRPairClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {7}",
|
||||
"",
|
||||
@@ -146,7 +147,6 @@
|
||||
"OpSize": "FEXCore::IR::OpSize",
|
||||
"SSA": "OrderedNode*",
|
||||
"GPR": "OrderedNode*",
|
||||
"GPRPair": "OrderedNode*",
|
||||
"FPR": "OrderedNode*",
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegisterClassType",
|
||||
@@ -168,7 +168,7 @@
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}": {
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}, i1:$ReadsParity{false}": {
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
@@ -245,24 +245,38 @@
|
||||
"Print SSA:$Value": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Debug operation that prints an SSA value to the console",
|
||||
"May only print 64bits of the value",
|
||||
"Depending on backend, may only support GPR printing"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) != GPRPairClass"
|
||||
]
|
||||
"May only print 64bits of the value"]
|
||||
},
|
||||
"GPRPair = RDRAND i1:$GetReseeded": {
|
||||
"GPR = AllocateGPR i1:$ForPair": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"Note: if an instruction uses allocated destinations-as-sources,",
|
||||
"it cannot use a regular destination too. This ensures RA correctness.",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"If ForPair is set, RA will try to allocate the base of a register pair"],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = AllocateFPR u8:#RegisterSize, u8:#ElementSize": {
|
||||
"Desc": ["Like AllocateGPR, but for FPR"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"GPR = AllocateGPRAfter GPR:$After": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"RA will attempt to allocate to the register after $After.",
|
||||
"It may not succeed."],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPR = RDRAND i1:$GetReseeded": {
|
||||
"Desc": ["Uses the hardware random number generator to generate a 64bit number",
|
||||
"The boolean argument asks if we should be reading the reseeded number or not",
|
||||
"Reseeded RNG calculation is more expensive and will be heavier to use",
|
||||
"The first GPR pair element is the 64-bit number",
|
||||
"The second GPR pair element is a bool if the number is valid",
|
||||
"Returns the 64-bit number",
|
||||
"Sets the Z flag if the number is valid.",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"Yield": {
|
||||
"HasSideEffects": true,
|
||||
@@ -277,10 +291,7 @@
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}, i1:$FromNZCV{false}": {
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Cmp1) == WalkFindRegClass($Cmp2)"
|
||||
]
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction GPR:$NewRIP": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
@@ -317,42 +328,18 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"GPRPair = CPUID GPR:$Function, GPR:$Leaf": {
|
||||
"Desc": ["Calls in to the CPUID handler function to return emulated CPUID",
|
||||
"Returns a 128bit GPR pair that fits emulated EAX, EBX, EDX, ECX respectively"
|
||||
],
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
"GPR:$EAX, GPR:$EBX, GPR:$ECX, GPR:$EDX = CPUID GPR:$Function, GPR:$Leaf": {
|
||||
"Desc": ["Calls in to the CPUID handler function to return emulated CPUID"],
|
||||
"DestSize": "4",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPRPair = XGetBV GPR:$Function": {
|
||||
"Desc": ["Calls in to the XCR handler function to return emulated XCR",
|
||||
"Returns a 64bit GPR pair that fits emulated EAX, EDX respectively"
|
||||
],
|
||||
"DestSize": "8",
|
||||
"NumElements": "2"
|
||||
"GPR:$EAX, GPR:$EDX = XGetBV GPR:$Function": {
|
||||
"Desc": ["Calls in to the XCR handler function to return emulated XCR"],
|
||||
"DestSize": "4",
|
||||
"HasSideEffects": true
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
"GPR = ExtractElementPair OpSize:#Size, GPRPair:$Pair, u8:$Element": {
|
||||
"Desc": ["Extracts a register for the register pair"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"GPRPair = CreateElementPair OpSize:#Size, GPR:$Lower, GPR:$Upper": {
|
||||
"Desc": ["Inserts a register for the register pair",
|
||||
"ssa0 is the lower incoming register",
|
||||
"ssa1 is the upper incoming register"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = Copy GPR:$Source": {
|
||||
"Desc": ["GPR copy, generated by RA to split live ranges"],
|
||||
"DestSize": "8"
|
||||
@@ -379,14 +366,36 @@
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = LoadPF u8:#Size": {
|
||||
"Desc": ["Loads raw PF"],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = LoadAF u8:#Size": {
|
||||
"Desc": ["Loads raw PF"],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreRegister SSA:$Value, u32:$Reg, RegisterClass:$Class, u8:#Size": {
|
||||
"HasSideEffects": true,
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores a value to a given register.",
|
||||
"Size must match the execution mode."],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
},
|
||||
|
||||
"StorePF GPR:$Value, u8:#Size": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores raw PF"],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreAF GPR:$Value, u8:#Size": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores raw AF"],
|
||||
"DestSize": "Size"
|
||||
}
|
||||
},
|
||||
"Memory": {
|
||||
@@ -403,6 +412,20 @@
|
||||
]
|
||||
},
|
||||
|
||||
"SSA:$Value1, SSA:$Value2 = LoadContextPair u8:#ByteSize, RegisterClass:$Class, u32:$Offset": {
|
||||
"Desc": ["Loads a pair of values from the context with offset",
|
||||
"Value0 = Ctx[Offset], Value1 = Ctx[Offset + ByteSize]"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContext u8:#ByteSize, RegisterClass:$Class, SSA:$Value, u32:$Offset": {
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
@@ -420,6 +443,24 @@
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContextPair u8:#ByteSize, RegisterClass:$Class, SSA:$Value1, SSA:$Value2, u32:$Offset": {
|
||||
"Desc": ["Stores a pair of values to the context with offset",
|
||||
"Ctx[Offset] = Value1, Ctx[Offset + ByteSize] = Value2",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
"SSA = LoadContextIndexed GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
|
||||
"Desc": ["Loads a value from the context with offset and indexed by SSA value",
|
||||
"Dest = Ctx[BaseOffset + Index * Stride]"
|
||||
@@ -493,6 +534,12 @@
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"SSA:$Value1, SSA:$Value2 = LoadMemPair RegisterClass:$Class, u8:#Size, GPR:$Addr, u32:$Offset": {
|
||||
"Desc": ["Load a pair of values from memory."],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"StoreMem RegisterClass:$Class, u8:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": [ "Stores a value to memory.",
|
||||
"Zero Extends if value's type is too small",
|
||||
@@ -505,6 +552,19 @@
|
||||
]
|
||||
},
|
||||
|
||||
"StoreMemPair RegisterClass:$Class, u8:#Size, SSA:$Value1, SSA:$Value2, GPR:$Addr, u32:$Offset": {
|
||||
"Desc": [ "Stores a pair of values to memory.",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class"
|
||||
]
|
||||
},
|
||||
|
||||
"SSA = LoadMemTSO RegisterClass:$Class, u8:#Size, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
@@ -598,6 +658,22 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = RMWHandle GPR:$Value": {
|
||||
"Desc": [
|
||||
"This is a special move that indicates the result will be poisoned by a non-SSA instruction writing to its result.",
|
||||
"In effect, it serves to prevent invalid optimizations with non-SSA instructions."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"TiedSource": 0
|
||||
},
|
||||
"GPR:$Addr, GPR:$Value = Pop u8:$Size, GPR:$Addr": {
|
||||
"Desc": [
|
||||
"Pops a value from the address, updating the new pointer after incrementing.",
|
||||
"The address is incremented by the size via an RMW source/destintaion."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = MemSet i1:$IsAtomic, u8:$Size, GPR:$Prefix, GPR:$Addr, GPR:$Value, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 STOS repeat",
|
||||
"Returns the final address that gets generated without the prefix appended."
|
||||
@@ -605,13 +681,12 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPRPair = MemCpy i1:$IsAtomic, u8:$Size, GPR:$Dest, GPR:$Src, GPR:$Length, GPR:$Direction": {
|
||||
"GPR:$DstAddress, GPR:$SrcAddress = MemCpy i1:$IsAtomic, u8:$Size, GPR:$Dest, GPR:$Src, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 MOVS repeat",
|
||||
"Returns the final addresses of destination and src addresses after they have been incremented or decremented"
|
||||
"Returns the final addresses after they have been incremented or decremented"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
"DestSize": "8"
|
||||
},
|
||||
"CacheLineClear GPR:$Addr, i1:$Serialize": {
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified",
|
||||
@@ -708,19 +783,17 @@
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPRPair = CASPair OpSize:#Size, GPRPair:$Expected, GPRPair:$Desired, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Does a compare and exchange with two GPRPair values",
|
||||
"GPR:$Lo, GPR:$Hi = CASPair OpSize:#Size, GPR:$ExpectedLo, GPR:$ExpectedHi, GPR:$DesiredLo, GPR:$DesiredHi, GPR:$Addr": {
|
||||
"Desc": ["Does a compare and exchange with two pairs of values",
|
||||
"ssa0 is the comparison value",
|
||||
"ssa1 is the new value",
|
||||
"ssa2 is the memory location",
|
||||
"Returns a pair containing the value in memory"
|
||||
"Returns the lower & upper halves of the value in memory."
|
||||
],
|
||||
"HasDest": true,
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicAdd OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
@@ -933,14 +1006,6 @@
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPRPair = TruncElementPair GPRPair:$Pair, u8:#ByteSize": {
|
||||
"Desc": [
|
||||
"Truncates each element of a pair to the destination size",
|
||||
"TODO: This IR op should get removed"
|
||||
],
|
||||
"DestSize": "ByteSize * 2",
|
||||
"NumElements": "2"
|
||||
},
|
||||
"GPR = CycleCounter": {
|
||||
"Desc": ["Returns the host 64bit cycle counter",
|
||||
"Useful when emulating rdtsc",
|
||||
@@ -1096,10 +1161,15 @@
|
||||
"Desc": ["Invert carry flag in NZCV"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"AXFlag": {
|
||||
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format"],
|
||||
"AXFlag GPR:$V_inv": {
|
||||
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format",
|
||||
"On FlagM2-less platforms, takes the inverted 1/0 overflow flag"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Parity GPR:$Raw, i1:$Mask, i1:$Invert": {
|
||||
"Desc": ["Calculates PF"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": {
|
||||
"Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"],
|
||||
"HasSideEffects": true
|
||||
@@ -1128,6 +1198,21 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcZero OpSize:#Size, GPR:$Src1": {
|
||||
"Desc": ["Adds GPR with inverted carry-in"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcZeroWithFlags OpSize:#Size, GPR:$Src1": {
|
||||
"Desc": ["Adds and set NZCV for the sum of GPR and inverted carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = SbbWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Subtracts and set NZCV for the difference of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
@@ -1179,10 +1264,11 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"CmpPairZ OpSize:#Size, GPRPair:$Src1, GPRPair:$Src2": {
|
||||
"CmpPairZ OpSize:#Size, GPR:$Src1Lo, GPR:$Src1Hi, GPR:$Src2Lo, GPR:$Src2Hi": {
|
||||
"Desc": ["Compares register pairs and sets Z accordingly, preserving N/Z/V.",
|
||||
"This accelerates cmpxchg."],
|
||||
"HasSideEffects": true
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs. ",
|
||||
@@ -1272,6 +1358,11 @@
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"TestZ OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the binary AND of two GPRs, setting Z accordingly and zeroing C and V. N is undefined."],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
],
|
||||
@@ -1296,12 +1387,16 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = ShiftFlags OpSize:$Size, GPR:$Result, GPR:$Src1, ShiftType:$Shift, GPR:$Src2, GPR:$PFInput": {
|
||||
"GPR = ShiftFlags OpSize:$Size, GPR:$Result, GPR:$Src1, ShiftType:$Shift, GPR:$Src2, GPR:$PFInput, i1:$InvertCF": {
|
||||
"Desc": ["Set NZCV flags for specified variable integer shift with given result.",
|
||||
"Returns updated raw PF."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"RotateFlags OpSize:$Size, GPR:$Result, GPR:$Shift, i1:$Left": {
|
||||
"Desc": ["Set NZCV flags for specified variable integer rotate with given result."],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Ror OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
],
|
||||
@@ -1445,6 +1540,16 @@
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = NZCVSelectIncrement OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Select and increment based on value in NZCV flags",
|
||||
"op:",
|
||||
"Dest = Cond ? TrueVal : (FalseVal + 1)"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"EmitValidation": [
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Select OpSize:#ResultSize, OpSize:$CompareSize, CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
@@ -1717,41 +1822,45 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (Vector1 * Vector2) + Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (Vector1 * Vector2) - Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFNMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFNMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (-Vector1 * Vector2) + Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFNMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFNMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (-Vector1 * Vector2) - Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
|
||||
@@ -52,7 +52,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {"EQ", "NEQ", "UGE", "ULT", "MI", "PL", "VS", "VC",
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "ANDZ", "ANDNZ",
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "TSTZ", "TSTNZ",
|
||||
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
|
||||
|
||||
*out << CondNames[Arg];
|
||||
@@ -77,8 +77,6 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else if (Arg == GPRPairClass.Val) {
|
||||
*out << "GPRPair";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
@@ -100,7 +98,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
@@ -316,7 +313,6 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
|
||||
@@ -38,7 +38,6 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case GPRClass:
|
||||
case GPRPairClass:
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
|
||||
@@ -71,7 +71,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass());
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
|
||||
@@ -18,7 +18,6 @@ class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass();
|
||||
|
||||
|
||||
@@ -16,7 +16,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <bit>
|
||||
@@ -53,17 +52,6 @@ static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
return ARMEmitter::Emitter::IsImmLogical(imm, width);
|
||||
}
|
||||
|
||||
static bool IsBfeAlreadyDone(IREmitter* IREmit, OrderedNodeWrapper src, uint64_t Width) {
|
||||
auto IROp = IREmit->GetOpHeader(src);
|
||||
if (IROp->Op == OP_BFE) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
if (Width >= Op->Width) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID)
|
||||
@@ -75,23 +63,40 @@ public:
|
||||
private:
|
||||
void HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR);
|
||||
void ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp);
|
||||
void ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR);
|
||||
|
||||
fextl::unordered_map<uint64_t, Ref> ConstPool;
|
||||
|
||||
// Pool inline constant generation. These are typically very small and pool efficiently.
|
||||
fextl::robin_map<uint64_t, Ref> InlineConstantGen;
|
||||
Ref CreateInlineConstant(IREmitter* IREmit, uint64_t Constant) {
|
||||
const auto it = InlineConstantGen.find(Constant);
|
||||
if (it != InlineConstantGen.end()) {
|
||||
return it->second;
|
||||
}
|
||||
auto Result = InlineConstantGen.insert_or_assign(Constant, IREmit->_InlineConstant(Constant));
|
||||
return Result.first->second;
|
||||
}
|
||||
bool SupportsTSOImm9 {};
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
|
||||
template<class F>
|
||||
bool InlineIf(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index, F Filter) {
|
||||
uint64_t Constant;
|
||||
if (!IREmit->IsValueConstant(IROp->Args[Index], &Constant) || !Filter(Constant)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[Index]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Index, IREmit->_InlineConstant(Constant));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Inline(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, [](uint64_t _) { return true; });
|
||||
}
|
||||
|
||||
bool InlineIfZero(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, [](uint64_t X) { return X == 0; });
|
||||
}
|
||||
|
||||
bool InlineIfLargeAddSub(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
auto Filter = [&IROp](uint64_t X) {
|
||||
return ARMEmitter::IsImmAddSub(X) && IROp->Size >= 4;
|
||||
};
|
||||
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
}
|
||||
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IROp_Header* IROp, OrderedNodeWrapper Offset,
|
||||
MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
uint64_t Imm {};
|
||||
@@ -112,7 +117,7 @@ private:
|
||||
|
||||
if (IsSIMM9 || IsExtended) {
|
||||
IREmit->SetWriteCursor(IR.GetNode(Offset));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Offset_Index, CreateInlineConstant(IREmit, Imm));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Offset_Index, IREmit->_InlineConstant(Imm));
|
||||
OffsetScale = 1;
|
||||
}
|
||||
}
|
||||
@@ -120,21 +125,66 @@ private:
|
||||
|
||||
// Constants are pooled per block.
|
||||
void ConstProp::HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
const uint32_t SSACount = CurrentIR.GetSSACount();
|
||||
|
||||
// Allocation/initialization deferred until first use, since many multiblocks
|
||||
// don't have constants leftover after all inlining.
|
||||
fextl::vector<Ref> Remap {};
|
||||
|
||||
struct Entry {
|
||||
int64_t Value;
|
||||
Ref R;
|
||||
};
|
||||
|
||||
|
||||
fextl::vector<Entry> Pool {};
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
Pool.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
auto it = ConstPool.find(Op->Constant);
|
||||
bool Found = false;
|
||||
|
||||
if (it != ConstPool.end()) {
|
||||
auto CodeIter = CurrentIR.at(CodeNode);
|
||||
IREmit->ReplaceUsesWithAfter(CodeNode, it->second, CodeIter);
|
||||
} else {
|
||||
ConstPool[Op->Constant] = CodeNode;
|
||||
// Search for the constant. This is O(n^2) but n is small since it's
|
||||
// local and most constants are inlined. In practice, it ends up much
|
||||
// faster than a hash table.
|
||||
for (auto K : Pool) {
|
||||
if (K.Value == Op->Constant) {
|
||||
uint32_t Value = CurrentIR.GetID(CodeNode).Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "def not yet remapped");
|
||||
|
||||
if (Remap.empty()) {
|
||||
Remap.resize(SSACount, nullptr);
|
||||
}
|
||||
|
||||
Remap[Value] = K.R;
|
||||
Found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!Found) {
|
||||
Pool.push_back({.Value = Op->Constant, .R = CodeNode});
|
||||
}
|
||||
} else if (!Remap.empty()) {
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t Value = IROp->Args[i].ID().Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "src not yet remapped");
|
||||
|
||||
Ref New = Remap[Value];
|
||||
if (New) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, New);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
ConstPool.clear();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -151,13 +201,25 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
bool IsConstant1 = IREmit->IsValueConstant(IROp->Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(IROp->Args[1], &Constant2);
|
||||
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
* here so we get the optimization for 32-bit adds too.
|
||||
*/
|
||||
if (Op->Header.Size == 4) {
|
||||
Constant1 = (int64_t)(int32_t)Constant1;
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
|
||||
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
break;
|
||||
} else if (IsConstant1 && IsConstant2 && IROp->Op == OP_SUB) {
|
||||
uint64_t NewConstant = (Constant1 - Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// So, negate the operation to negate (and inline) the constant.
|
||||
if (IROp->Op == OP_ADD) {
|
||||
@@ -178,6 +240,23 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
// Replace the second source with the negated constant.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant);
|
||||
}
|
||||
|
||||
if (!InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1) && (IROp->Op == OP_SUB || IROp->Op == OP_SUBWITHFLAGS)) {
|
||||
// TODO: Generalize this
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ADDNZCV: {
|
||||
InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_SUBNZCV: {
|
||||
if (!InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
// TODO: Generalize this
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
@@ -197,50 +276,38 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
bool Replaced = false;
|
||||
|
||||
// Order matter for short circuit evaluation, subsequent ifs read constant2.
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
uint64_t NewConstant = (Constant1 & Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (Constant2 == 1) {
|
||||
// happens from flag calcs
|
||||
auto val = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
|
||||
uint64_t Constant3;
|
||||
if (val->Op == OP_SELECT && IREmit->IsValueConstant(val->Args[2], &Constant2) && IREmit->IsValueConstant(val->Args[3], &Constant3) &&
|
||||
Constant2 == 1 && Constant3 == 0) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
Replaced = true;
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
Replaced = true;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | Constant2;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
// OR with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
break;
|
||||
}
|
||||
case OP_XOR: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 ^ Constant2;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
// XOR with same value results to zero
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, IREmit->_Constant(0));
|
||||
} else {
|
||||
// XOR with zero results in the nonzero source
|
||||
bool Replaced = false;
|
||||
for (unsigned i = 0; i < 2; ++i) {
|
||||
if (!IREmit->IsValueConstant(IROp->Args[i], &Constant1)) {
|
||||
continue;
|
||||
@@ -253,11 +320,22 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[1 - i]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
Replaced = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN:
|
||||
case OP_TESTNZ: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
break;
|
||||
}
|
||||
case OP_NEG: {
|
||||
uint64_t Constant {};
|
||||
|
||||
@@ -267,6 +345,11 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ASHR:
|
||||
case OP_ROR: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHL: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
@@ -280,26 +363,20 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_LSHR: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
// The source is masked, which will produce a correctly masked
|
||||
// destination. Masking the destination without the source instead will
|
||||
// right-shift garbage into the upper bits instead of zeroes.
|
||||
Constant1 &= getMask(IROp);
|
||||
Constant2 &= (IROp->Size == 8 ? 63 : 31);
|
||||
uint64_t NewConstant = (Constant1 >> Constant2);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -307,46 +384,12 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
|
||||
// Is this value already BFE'd?
|
||||
if (IsBfeAlreadyDone(IREmit, Op->Src, Op->Width)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Src));
|
||||
break;
|
||||
}
|
||||
|
||||
// Is this value already ZEXT'd?
|
||||
if (Op->lsb == 0) {
|
||||
// LoadMem, LoadMemTSO & LoadContext ZExt
|
||||
auto source = Op->Src;
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (Op->Width >= (sourceHeader->Size * 8) &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)) {
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (IROp->Size <= 8 && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
uint64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IROp->Size == CurrentIR.GetOp<IROp_Header>(IROp->Args[0])->Size && Op->Width == (IROp->Size * 8) && Op->lsb == 0) {
|
||||
// A BFE that extracts all bits results in original value
|
||||
// XXX - This is broken for now - see https://github.com/FEX-Emu/FEX/issues/351
|
||||
// IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
} else if (Op->Width == 1 && Op->lsb == 0) {
|
||||
// common from flag codegen
|
||||
auto val = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (val->Op == OP_SELECT && IREmit->IsValueConstant(val->Args[2], &Constant2) && IREmit->IsValueConstant(val->Args[3], &Constant3) &&
|
||||
Constant2 == 1 && Constant3 == 0) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
@@ -371,18 +414,10 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantDest {};
|
||||
uint64_t ConstantSrc {};
|
||||
bool DestIsConstant = IREmit->IsValueConstant(IROp->Args[0], &ConstantDest);
|
||||
bool SrcIsConstant = IREmit->IsValueConstant(IROp->Args[1], &ConstantSrc);
|
||||
|
||||
if (DestIsConstant && SrcIsConstant) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t NewConstant = ConstantDest & ~(SourceMask << Op->lsb);
|
||||
NewConstant |= (ConstantSrc & SourceMask) << Op->lsb;
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) {
|
||||
if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) {
|
||||
// We are trying to insert constant, if it is a bitfield of only set bits then we can orr or and it.
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
@@ -399,24 +434,6 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MUL: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 * Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && std::popcount(Constant2) == 1) {
|
||||
if (IROp->Size == 4 || IROp->Size == 8) {
|
||||
uint64_t amt = std::countr_zero(Constant2);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto shift = IREmit->_Lshl(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(amt));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, shift);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
@@ -483,14 +500,14 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
// If the CPUID needs a constant leaf to be optimized then this can't work if we didn't const-prop the leaf register.
|
||||
if (!(SupportsConstant.NeedsLeaf == CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT && !IsConstantLeaf)) {
|
||||
// Calculate the constant data and replace all uses.
|
||||
// DCE will remove the CPUID IR operation.
|
||||
const auto ConstantCPUIDResult = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
uint64_t ResultsLower = (static_cast<uint64_t>(ConstantCPUIDResult.ebx) << 32) | ConstantCPUIDResult.eax;
|
||||
uint64_t ResultsUpper = (static_cast<uint64_t>(ConstantCPUIDResult.edx) << 32) | ConstantCPUIDResult.ecx;
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair = IREmit->_CreateElementPair(IR::OpSize::i128Bit, IREmit->_Constant(ResultsLower), IREmit->_Constant(ResultsUpper));
|
||||
// Replace all CPUID uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEBX), IREmit->_Constant(Result.ebx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutECX), IREmit->_Constant(Result.ecx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -502,12 +519,11 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
uint64_t ConstantFunction {};
|
||||
if (IREmit->IsValueConstant(Op->Function, &ConstantFunction) && CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto ConstantXCRResult = CPUID->RunXCRFunction(ConstantFunction);
|
||||
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair =
|
||||
IREmit->_CreateElementPair(IR::OpSize::i64Bit, IREmit->_Constant(ConstantXCRResult.eax), IREmit->_Constant(ConstantXCRResult.edx));
|
||||
// Replace all xgetbv uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -564,246 +580,101 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
break;
|
||||
}
|
||||
|
||||
default: break;
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT:
|
||||
case OP_RMIFNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
|
||||
void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
InlineConstantGen.clear();
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
switch (IROp->Op) {
|
||||
case OP_LSHR:
|
||||
case OP_ASHR:
|
||||
case OP_ROR:
|
||||
case OP_LSHL: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[2], &Constant2) && IREmit->IsValueConstant(IROp->Args[3], &Constant3) &&
|
||||
(Constant2 == 1 || Constant2 == AllOnes) && Constant3 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[2]));
|
||||
|
||||
// this shouldn't be here, but rather on the emitter themselves or the constprop transformation?
|
||||
if (IROp->Size <= 4) {
|
||||
Constant2 &= 31;
|
||||
} else {
|
||||
Constant2 &= 63;
|
||||
}
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
}
|
||||
break;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
case OP_ADDNZCV:
|
||||
case OP_SUBNZCV:
|
||||
case OP_ADDWITHFLAGS:
|
||||
case OP_SUBWITHFLAGS: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
if (ARMEmitter::IsImmAddSub(Constant2) && IROp->Size >= 4) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
}
|
||||
} else if (IROp->Op == OP_SUBNZCV || IROp->Op == OP_SUBWITHFLAGS || IROp->Op == OP_SUB) {
|
||||
// TODO: Generalize this
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_RMIFNZCV: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1)) {
|
||||
if (IsImmLogical(Constant1, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT: {
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
if (InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[2], &Constant2) && IREmit->IsValueConstant(IROp->Args[3], &Constant3) &&
|
||||
(Constant2 == 1 || Constant2 == AllOnes) && Constant3 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, CreateInlineConstant(IREmit, Constant3));
|
||||
}
|
||||
|
||||
break;
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 0, [&AllOnes](uint64_t X) { return X == 1 || X == AllOnes; });
|
||||
}
|
||||
case OP_NZCVSELECT: {
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
break;
|
||||
}
|
||||
case OP_EXITFUNCTION: {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
uint64_t Constant0 {};
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1) && Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant0) && (Constant0 == 1 || Constant0 == AllOnes)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_EXITFUNCTION: {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) {
|
||||
if (!Inline(IREmit, CurrentIR, CodeNode, IROp, Op->NewRIP_Index)) {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant));
|
||||
} else {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(IR::SizeToOpSize(EO->Header.Size), EO->Offset));
|
||||
}
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(IR::SizeToOpSize(EO->Header.Size), EO->Offset));
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_AND:
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN: {
|
||||
uint64_t Constant2 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (IsImmLogical(Constant2, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
break;
|
||||
}
|
||||
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -812,13 +683,11 @@ void ConstProp::Run(IREmitter* IREmit) {
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
HandleConstantPools(IREmit, CurrentIR);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
}
|
||||
|
||||
ConstantInlining(IREmit, CurrentIR);
|
||||
HandleConstantPools(IREmit, IREmit->ViewIR());
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID) {
|
||||
|
||||
@@ -1,144 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Cross block store-after-store elimination
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
constexpr int PropagationRounds = 5;
|
||||
|
||||
// Return a bit representing a single GPR or FPR.
|
||||
static inline uint64_t RegBit(RegisterClassType Class, uint32_t Reg) {
|
||||
uint32_t AdjustedReg = (Class == FPRClass) ? (32 + Reg) : Reg;
|
||||
|
||||
return 1UL << AdjustedReg;
|
||||
}
|
||||
|
||||
class DeadStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
void Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
struct ReadWriteKill {
|
||||
uint64_t reads {0};
|
||||
uint64_t writes {0};
|
||||
uint64_t kill {0};
|
||||
};
|
||||
|
||||
struct Info {
|
||||
ReadWriteKill reg;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead reg stores
|
||||
*
|
||||
* First pass computes which regs are read and written per block
|
||||
*
|
||||
* Second pass computes which regs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead regs across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
void DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::vector<Info> InfoMap(CurrentIR.GetSSACount());
|
||||
|
||||
// Pass 1
|
||||
// Compute regs read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
BlockInfo.reg.writes |= RegBit(Op->Class, Op->Reg);
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
BlockInfo.reg.reads |= RegBit(Op->Class, Op->Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute flags/registers that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < PropagationRounds; i++) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto CodeBlock = BlockIROp->C<IROp_CodeBlock>();
|
||||
|
||||
auto IROp = CurrentIR.GetNode(CurrentIR.GetNode(CodeBlock->Last)->Header.Previous)->Op(CurrentIR.GetData());
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
auto& TargetInfo = InfoMap[Op->Header.Args[0].ID().Value];
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.reg.kill = TargetInfo.reg.writes & ~(TargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
// If written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
|
||||
} else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
auto& TrueTargetInfo = InfoMap[Op->TrueBlock.ID().Value];
|
||||
auto& FalseTargetInfo = InfoMap[Op->FalseBlock.ID().Value];
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.reg.kill = TrueTargetInfo.reg.writes & ~(TrueTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
BlockInfo.reg.kill &= FalseTargetInfo.reg.writes & ~(FalseTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
// if written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.reg.kill & RegBit(Op->Class, Op->Reg)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination() {
|
||||
return fextl::make_unique<DeadStoreElimination>();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -22,13 +22,11 @@ namespace FEXCore::IR::Validation {
|
||||
struct RegState {
|
||||
static constexpr IR::NodeID UninitializedValue {0};
|
||||
static constexpr IR::NodeID InvalidReg {0xffff'ffff};
|
||||
static constexpr IR::NodeID CorruptedPair {0xffff'fffe};
|
||||
|
||||
// This class makes some assumptions about how the host registers are arranged and mapped to virtual registers:
|
||||
// 1. There will be less than 32 GPRs and 32 FPRs
|
||||
// 2. If the GPRFixed class is used, there will be 16 GPRs and 16 FixedGPRs max
|
||||
// 3. Same with FPRFixed
|
||||
// 4. If the GPRPairClass is used, it is assumed each GPRPair N will map onto GPRs N and N + 1
|
||||
|
||||
// These assumptions were all true for the state of the arm64 and x86 jits at the time this was written
|
||||
|
||||
@@ -49,11 +47,6 @@ struct RegState {
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case GPRPairClass:
|
||||
// Alias paired registers onto both
|
||||
GPRs[Reg.Reg] = ssa;
|
||||
GPRs[Reg.Reg + 1] = ssa;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -66,12 +59,6 @@ struct RegState {
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
case GPRPairClass:
|
||||
// Make sure both halves of the Pair contain the same SSA
|
||||
if (GPRs[Reg.Reg] == GPRs[Reg.Reg + 1]) {
|
||||
return GPRs[Reg.Reg];
|
||||
}
|
||||
return CorruptedPair;
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
@@ -139,14 +126,6 @@ void RAValidation::Run(IREmitter* IREmit) {
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::CorruptedPair) {
|
||||
HadError |= true;
|
||||
|
||||
auto Lower = BlockRegState.Get(PhysicalRegister(GPRClass, uint8_t(PhyReg.Reg * 2) + 1));
|
||||
auto Upper = BlockRegState.Get(PhysicalRegister(GPRClass, PhyReg.Reg * 2 + 1));
|
||||
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects paired reg{} to contain %{}, but it actually contains {{%{}, %{}}}\n", ID, i,
|
||||
PhyReg.Reg, ArgID, Lower, Upper);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
|
||||
@@ -7,6 +7,10 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/fextl/deque.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
@@ -14,16 +18,13 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
|
||||
// Flag bit flags
|
||||
#define FLAG_V (1U << 0)
|
||||
#define FLAG_C (1U << 1)
|
||||
#define FLAG_Z (1U << 2)
|
||||
#define FLAG_N (1U << 3)
|
||||
#define FLAG_A (1U << 4)
|
||||
#define FLAG_P (1U << 5)
|
||||
#define FLAG_P (1U << 4)
|
||||
#define FLAG_A (1U << 5)
|
||||
|
||||
#define FLAG_ZCV (FLAG_Z | FLAG_C | FLAG_V)
|
||||
#define FLAG_NZCV (FLAG_N | FLAG_ZCV)
|
||||
@@ -31,10 +32,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
struct FlagInfo {
|
||||
// If set, all following fields are zero, used for a quick exit.
|
||||
bool Trivial;
|
||||
|
||||
struct FlagInfoUnpacked {
|
||||
// Set of flags read by the instruction.
|
||||
unsigned Read;
|
||||
|
||||
@@ -45,10 +43,106 @@ struct FlagInfo {
|
||||
// eliminated.
|
||||
bool CanEliminate;
|
||||
|
||||
// If true, the opcode can be replaced with Replacement if its flag writes can
|
||||
// all be eliminated.
|
||||
bool CanReplace;
|
||||
// If set, the opcode can be replaced with Replacement if its flag writes can
|
||||
// all be eliminated, or ReplacementNoWrite if its register write can be
|
||||
// eliminated.
|
||||
IROps Replacement;
|
||||
IROps ReplacementNoWrite;
|
||||
|
||||
// Needs speical handling
|
||||
bool Special;
|
||||
};
|
||||
|
||||
struct FlagInfo {
|
||||
uint64_t Raw;
|
||||
|
||||
static constexpr struct FlagInfo Pack(struct FlagInfoUnpacked F) {
|
||||
uint64_t R = F.Read | (F.Write << 8) | (F.CanEliminate << 16) | (((uint64_t)F.Replacement) << 32) |
|
||||
((uint64_t)F.ReplacementNoWrite << 48) | (F.Special ? (1ull << 63) : 0);
|
||||
return {.Raw = R};
|
||||
}
|
||||
|
||||
bool Trivial() {
|
||||
return Raw == 0;
|
||||
}
|
||||
|
||||
unsigned Read() {
|
||||
return Bits(0, 8);
|
||||
}
|
||||
|
||||
unsigned Write() {
|
||||
return Bits(8, 8);
|
||||
}
|
||||
|
||||
bool CanEliminate() {
|
||||
return Bits(16, 1);
|
||||
}
|
||||
|
||||
bool Special() {
|
||||
return Bits(63, 1);
|
||||
}
|
||||
|
||||
IROps Replacement() {
|
||||
return (IROps)Bits(32, 16);
|
||||
}
|
||||
|
||||
IROps ReplacementNoWrite() {
|
||||
return (IROps)Bits(48, 16);
|
||||
}
|
||||
|
||||
private:
|
||||
unsigned Bits(unsigned Start, unsigned Count) {
|
||||
return (Raw >> Start) & ((1u << Count) - 1);
|
||||
}
|
||||
};
|
||||
|
||||
struct BlockInfo {
|
||||
fextl::vector<Ref> Predecessors;
|
||||
uint8_t Flags;
|
||||
bool InWorklist;
|
||||
};
|
||||
|
||||
struct ControlFlowGraph {
|
||||
fextl::unordered_map<uint32_t, BlockInfo> BlockMap;
|
||||
IRListView& IR;
|
||||
|
||||
void AddBlock(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
uint32_t ID = IR.GetID(Block).Value;
|
||||
|
||||
// Add the block with conservative flags and already in the worklist.
|
||||
auto Info = &BlockMap.emplace(ID, BlockInfo {{}, FLAG_ALL, true}).first->second;
|
||||
|
||||
// Add some initial capacity
|
||||
Info->Predecessors.reserve(2);
|
||||
|
||||
// Add to worklist
|
||||
Worklist.push_back(Block);
|
||||
}
|
||||
|
||||
BlockInfo* Get(uint32_t Block) {
|
||||
return &BlockMap.try_emplace(Block).first->second;
|
||||
}
|
||||
|
||||
BlockInfo* Get(Ref Block) {
|
||||
return Get(IR.GetID(Block).Value);
|
||||
}
|
||||
|
||||
BlockInfo* Get(OrderedNodeWrapper Block) {
|
||||
return Get(Block.ID().Value);
|
||||
}
|
||||
|
||||
void RecordEdge(Ref From, Ref To) {
|
||||
auto Info = Get(To);
|
||||
Info->Predecessors.push_back(From);
|
||||
}
|
||||
|
||||
void AddWorklist(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
auto Info = Get(Block);
|
||||
if (!Info->InWorklist) {
|
||||
Info->InWorklist = true;
|
||||
Worklist.push_front(Block);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass {
|
||||
@@ -60,10 +154,10 @@ private:
|
||||
unsigned FlagForReg(unsigned Reg);
|
||||
unsigned FlagsForCondClassType(CondClassType Cond);
|
||||
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagForReg(unsigned Reg) {
|
||||
return Reg == Core::CPUState::PF_AS_GREG ? FLAG_P : Reg == Core::CPUState::AF_AS_GREG ? FLAG_A : 0;
|
||||
void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode);
|
||||
CondClassType X86ToArmFloatCond(CondClassType X86);
|
||||
bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG);
|
||||
void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
|
||||
@@ -101,133 +195,186 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType C
|
||||
}
|
||||
}
|
||||
|
||||
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
constexpr FlagInfo ClassifyConst(IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_ANDWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_AND,
|
||||
};
|
||||
.ReplacementNoWrite = OP_TESTNZ,
|
||||
});
|
||||
|
||||
case OP_ADDWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADD,
|
||||
};
|
||||
.ReplacementNoWrite = OP_ADDNZCV,
|
||||
});
|
||||
|
||||
case OP_SUBWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_SUB,
|
||||
};
|
||||
.ReplacementNoWrite = OP_SUBNZCV,
|
||||
});
|
||||
|
||||
case OP_ADCWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADC,
|
||||
};
|
||||
.ReplacementNoWrite = OP_ADCNZCV,
|
||||
});
|
||||
|
||||
case OP_ADCZEROWITHFLAGS:
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.Replacement = OP_ADCZERO,
|
||||
});
|
||||
|
||||
case OP_SBBWITHFLAGS:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_SBB,
|
||||
};
|
||||
.ReplacementNoWrite = OP_SBBNZCV,
|
||||
});
|
||||
|
||||
case OP_SHIFTFLAGS:
|
||||
// _ShiftFlags conditionally sets NZCV+PF, which we model here as a
|
||||
// read-modify-write. Logically, it also conditionally makes AF undefined,
|
||||
// which we model by omitting AF from both Read and Write sets (since
|
||||
// "cond ? AF : undef" may be optimized to "AF").
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_NZCV | FLAG_P,
|
||||
.Write = FLAG_NZCV | FLAG_P,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_ROTATEFLAGS:
|
||||
// _RotateFlags conditionally sets CV, again modeled as RMW.
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C | FLAG_V,
|
||||
.Write = FLAG_C | FLAG_V,
|
||||
.CanEliminate = true,
|
||||
});
|
||||
|
||||
case OP_RDRAND: return FlagInfo::Pack({.Write = FLAG_NZCV});
|
||||
|
||||
case OP_ADDNZCV:
|
||||
case OP_SUBNZCV:
|
||||
case OP_TESTNZ:
|
||||
case OP_FCMP:
|
||||
case OP_STORENZCV:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_AXFLAG:
|
||||
// Per the Arm spec, axflag reads Z/V/C but not N. It writes all flags.
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_ZCV,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_CMPPAIRZ:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_Z,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_CARRYINVERT:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_C,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_SETSMALLNZV:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = FLAG_N | FLAG_Z | FLAG_V,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_LOADNZCV: return {.Read = FLAG_NZCV};
|
||||
case OP_LOADNZCV: return FlagInfo::Pack({.Read = FLAG_NZCV});
|
||||
|
||||
case OP_ADC:
|
||||
case OP_SBB: return {.Read = FLAG_C};
|
||||
case OP_ADCZERO:
|
||||
case OP_SBB: return FlagInfo::Pack({.Read = FLAG_C});
|
||||
|
||||
case OP_ADCNZCV:
|
||||
case OP_SBBNZCV:
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
|
||||
case OP_NZCVSELECT: {
|
||||
case OP_LOADPF: return FlagInfo::Pack({.Read = FLAG_P});
|
||||
case OP_LOADAF: return FlagInfo::Pack({.Read = FLAG_A});
|
||||
case OP_STOREPF: return FlagInfo::Pack({.Write = FLAG_P, .CanEliminate = true});
|
||||
case OP_STOREAF: return FlagInfo::Pack({.Write = FLAG_A, .CanEliminate = true});
|
||||
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTINCREMENT:
|
||||
case OP_NEG:
|
||||
case OP_CONDJUMP:
|
||||
case OP_CONDSUBNZCV:
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_RMIFNZCV:
|
||||
case OP_INVALIDATEFLAGS: return FlagInfo::Pack({.Special = true});
|
||||
default: return FlagInfo::Pack({});
|
||||
}
|
||||
}
|
||||
|
||||
constexpr auto FlagInfos = std::invoke([] {
|
||||
std::array<FlagInfo, OP_LAST> ret = {};
|
||||
|
||||
for (unsigned i = 0; i < OP_LAST; ++i) {
|
||||
ret[i] = ClassifyConst((IROps)i);
|
||||
}
|
||||
|
||||
return ret;
|
||||
});
|
||||
|
||||
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
FlagInfo Info = FlagInfos[IROp->Op];
|
||||
if (!Info.Special()) {
|
||||
return Info;
|
||||
}
|
||||
|
||||
switch (IROp->Op) {
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTINCREMENT: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NEG: {
|
||||
auto Op = IROp->CW<IR::IROp_Neg>();
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
if (!Op->FromNZCV) {
|
||||
break;
|
||||
return FlagInfo::Pack({});
|
||||
}
|
||||
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_CONDSUBNZCV:
|
||||
case OP_CONDADDNZCV: {
|
||||
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Read = FlagsForCondClassType(Op->Cond),
|
||||
.Write = FLAG_NZCV,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
case OP_RMIFNZCV: {
|
||||
@@ -238,10 +385,10 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
static_assert(FLAG_C == (1 << 1), "rmif mask lines up with our bits");
|
||||
static_assert(FLAG_V == (1 << 0), "rmif mask lines up with our bits");
|
||||
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = Op->Mask,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
case OP_INVALIDATEFLAGS: {
|
||||
@@ -276,39 +423,16 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
// The mental model of InvalidateFlags is writing undefined values to all
|
||||
// of the selected flags, allowing the write-after-write optimizations to
|
||||
// optimize invalidate-after-write for free.
|
||||
return {
|
||||
return FlagInfo::Pack({
|
||||
.Write = Flags,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
case OP_LOADREGISTER: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadRegister>();
|
||||
if (Op->Class != GPRClass) {
|
||||
break;
|
||||
}
|
||||
|
||||
return {.Read = FlagForReg(Op->Reg)};
|
||||
default: LOGMAN_THROW_AA_FMT(false, "invalid special op"); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
case OP_STOREREGISTER: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreRegister>();
|
||||
if (Op->Class != GPRClass) {
|
||||
break;
|
||||
}
|
||||
|
||||
unsigned Flag = FlagForReg(Op->Reg);
|
||||
|
||||
return {
|
||||
.Write = Flag,
|
||||
.CanEliminate = Flag != 0,
|
||||
};
|
||||
}
|
||||
|
||||
default: break;
|
||||
}
|
||||
|
||||
return {.Trivial = true};
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// General purpose dead code elimination. Returns whether flag handling should
|
||||
@@ -352,79 +476,295 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref
|
||||
return true;
|
||||
}
|
||||
|
||||
CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) {
|
||||
// Table of x86 condition codes that map to arm64 condition codes, in the
|
||||
// sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm).
|
||||
//
|
||||
// E would be "equal or unordered", no condition code.
|
||||
// G would be "greater than or less than", no condition code.
|
||||
//
|
||||
// SF/OF conditions are trivial and therefore shouldn't actually be generated
|
||||
switch (X86) {
|
||||
case COND_UGE /* A */: return {COND_FGE} /* GE */;
|
||||
case COND_UGT /* AE */: return {COND_FGT} /* GT */;
|
||||
case COND_ULT /* B */: return {COND_SLT} /* LT */;
|
||||
case COND_ULE /* BE */: return {COND_SLE} /* LE */;
|
||||
case COND_SLE /* LE */: return {COND_SLE} /* LE */;
|
||||
default: return {COND_AL};
|
||||
}
|
||||
}
|
||||
|
||||
void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode) {
|
||||
// Skip past StoreRegisters at the end -- they don't touch flags.
|
||||
auto PrevWrap = CodeNode->Header.Previous;
|
||||
while (CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREREGISTER ||
|
||||
CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREPF || CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREAF) {
|
||||
PrevWrap = CurrentIR.GetNode(PrevWrap)->Header.Previous;
|
||||
}
|
||||
|
||||
auto Prev = CurrentIR.GetOp<IR::IROp_Header>(PrevWrap);
|
||||
if (Prev->Op == OP_AXFLAG) {
|
||||
// Pattern match a branch fed by AXFLAG.
|
||||
CondClassType ArmCond = X86ToArmFloatCond(Op->Cond);
|
||||
if (ArmCond == COND_AL) {
|
||||
return;
|
||||
}
|
||||
|
||||
Op->Cond = ArmCond;
|
||||
} else if (Prev->Op == OP_SUBNZCV) {
|
||||
// Pattern match a branch fed by a compare. We could also handle bit tests
|
||||
// here, but tbz/tbnz has a limited offset range which we don't have a way to
|
||||
// deal with yet. Let's hope that's not a big deal.
|
||||
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < 4)) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto SecondArg = CurrentIR.GetOp<IR::IROp_Header>(Prev->Args[1]);
|
||||
if (SecondArg->Op != OP_INLINECONSTANT || SecondArg->C<IR::IROp_InlineConstant>()->Constant != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We've matched. Fold the compare into branch.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CurrentIR.GetNode(Prev->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CurrentIR.GetNode(Prev->Args[1]));
|
||||
Op->FromNZCV = false;
|
||||
Op->CompareSize = Prev->Size;
|
||||
} else {
|
||||
return;
|
||||
}
|
||||
|
||||
// The compare/test/axflag sets flags but does not write registers. Flags are
|
||||
// dead after the jump. The jump does not read flags anymore. There is no
|
||||
// intervening instruction. Therefore the compare is dead.
|
||||
IREmit->Remove(CurrentIR.GetNode(PrevWrap));
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief This pass removes dead code locally.
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG) {
|
||||
uint32_t FlagsRead = FLAG_ALL;
|
||||
|
||||
// Reverse iteration is not yet working with the iterators
|
||||
auto BlockIROp = CurrentIR.GetOp<IR::IROp_CodeBlock>(Block);
|
||||
|
||||
// We grab these nodes this way so we can iterate easily
|
||||
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
|
||||
// Advance past EndBlock to get at the exit.
|
||||
--CodeLast;
|
||||
|
||||
// Initialize the FlagsRead mask according to the exit instruction.
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
FlagsRead = CFG.Get(ExitOp->Args[0])->Flags;
|
||||
}
|
||||
|
||||
// Iterate the block in reverse
|
||||
while (true) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
// Optimizing flags can cause earlier flag reads to become dead but dead
|
||||
// flag reads should not impede optimiation of earlier dead flag writes.
|
||||
// We must DCE as we go to ensure we converge in a single iteration.
|
||||
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
|
||||
// Optimiation algorithm: For each flag written...
|
||||
//
|
||||
// If the flag has a later read (per FlagsRead), remove the flag from
|
||||
// FlagsRead, since the reader is covered by this write.
|
||||
//
|
||||
// Else, there is no later read, so remove the flag write (if we can).
|
||||
// This is the active part of the optimization.
|
||||
//
|
||||
// Then, add each flag read to FlagsRead.
|
||||
//
|
||||
// This order is important: instructions that read-modify-write flags
|
||||
// (like adcs) first read flags, then write flags. Since we're iterating
|
||||
// the block backwards, that means we handle the write first.
|
||||
struct FlagInfo Info = Classify(IROp);
|
||||
|
||||
if (!Info.Trivial()) {
|
||||
bool Eliminated = false;
|
||||
|
||||
if ((FlagsRead & Info.Write()) == 0) {
|
||||
if ((Info.CanEliminate() || Info.Replacement()) && CodeNode->GetUses() == 0) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Eliminated = true;
|
||||
} else if (Info.Replacement()) {
|
||||
IROp->Op = Info.Replacement();
|
||||
}
|
||||
} else if (Info.ReplacementNoWrite() && CodeNode->GetUses() == 0) {
|
||||
IROp->Op = Info.ReplacementNoWrite();
|
||||
}
|
||||
|
||||
// If we don't care about the sign or carry, we can optimize testnz.
|
||||
// Carry is inverted between testz and testnz so we check that too. Note
|
||||
// this flag is outside of the if, since the TestNZ might result from
|
||||
// optimizing AndWithFlags, and we need to converge locally in a single
|
||||
// iteration.
|
||||
if (IROp->Op == OP_TESTNZ && IROp->Size < 4 && !(FlagsRead & (FLAG_N | FLAG_C))) {
|
||||
IROp->Op = OP_TESTZ;
|
||||
}
|
||||
|
||||
FlagsRead &= ~Info.Write();
|
||||
|
||||
// If we eliminated the instruction, we eliminate its read too. This
|
||||
// check is required to ensure the pass converges locally in a single
|
||||
// iteration.
|
||||
if (!Eliminated) {
|
||||
FlagsRead |= Info.Read();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Iterate in reverse
|
||||
if (CodeLast == CodeBegin) {
|
||||
break;
|
||||
}
|
||||
--CodeLast;
|
||||
}
|
||||
|
||||
// For the purposes of global propagation, the content of our progress doesn't
|
||||
// matter -- only the difference in our final FlagsRead contributes to changes
|
||||
// in the predecessors.
|
||||
uint32_t OldFlagsRead = CFG.Get(Block)->Flags;
|
||||
CFG.Get(Block)->Flags = FlagsRead;
|
||||
return (OldFlagsRead != FlagsRead);
|
||||
}
|
||||
|
||||
void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG) {
|
||||
// Mapping for flags inside this pass.
|
||||
const uint8_t PARTIAL = 0;
|
||||
const uint8_t FULL = 1;
|
||||
|
||||
// Initialize conservatively: all blocks need full parity. This initialization
|
||||
// matters for proper handling of backedges.
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
CFG.Get(Block)->Flags = FULL;
|
||||
}
|
||||
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
bool Full = false;
|
||||
auto Predecessors = CFG.Get(Block)->Predecessors;
|
||||
|
||||
if (Predecessors.empty()) {
|
||||
// Conservatively assume there was full parity before the start block
|
||||
Full = true;
|
||||
} else {
|
||||
// If any predecessor needs full parity at the end, we need full parity.
|
||||
for (auto Pred : Predecessors) {
|
||||
Full |= (CFG.Get(Pred)->Flags == FULL);
|
||||
}
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) {
|
||||
if (IROp->Op == OP_STOREPF) {
|
||||
auto Op = IROp->CW<IR::IROp_StorePF>();
|
||||
auto Generator = CurrentIR.GetOp<IR::IROp_Header>(Op->Value);
|
||||
|
||||
// Determine if we only write 0/1 to the parity flag.
|
||||
Full = true;
|
||||
if (Generator->Op == OP_NZCVSELECT) {
|
||||
auto C0 = CurrentIR.GetOp<IR::IROp_Header>(Generator->Args[0]);
|
||||
auto C1 = CurrentIR.GetOp<IR::IROp_Header>(Generator->Args[1]);
|
||||
if (C0->Op == C1->Op && C0->Op == OP_INLINECONSTANT) {
|
||||
auto IC0 = CurrentIR.GetOp<IR::IROp_InlineConstant>(Generator->Args[0]);
|
||||
auto IC1 = CurrentIR.GetOp<IR::IROp_InlineConstant>(Generator->Args[1]);
|
||||
|
||||
// We need the full 8 if the constant has upper bits set.
|
||||
Full = (IC0->Constant | IC1->Constant) & ~1;
|
||||
}
|
||||
}
|
||||
} else if (IROp->Op == OP_PARITY && !Full) {
|
||||
// Eliminate parity calculations if it's only 1-bit.
|
||||
auto Parity = IROp->C<IROp_Parity>();
|
||||
Ref Value = CurrentIR.GetNode(Parity->Raw);
|
||||
|
||||
if (Parity->Invert) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Value = IREmit->_Xor(OpSize::i32Bit, Value, IREmit->_InlineConstant(1));
|
||||
}
|
||||
|
||||
IREmit->ReplaceUsesWithAfter(CodeNode, Value, CurrentIR.at(CodeNode));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
|
||||
// Record our final state for our successors to read.
|
||||
CFG.Get(Block)->Flags = Full ? FULL : PARTIAL;
|
||||
}
|
||||
}
|
||||
|
||||
void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::deque<Ref> Worklist;
|
||||
|
||||
ControlFlowGraph CFG {.IR = CurrentIR};
|
||||
|
||||
// Gather blocks
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
// We model all flags as read at the end of the block, since this pass is
|
||||
// presently purely local. Optimizing this requires global anslysis.
|
||||
uint32_t FlagsRead = FLAG_ALL;
|
||||
CFG.AddBlock(Worklist, BlockNode);
|
||||
}
|
||||
|
||||
// Reverse iteration is not yet working with the iterators
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
// Gather CFG
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto CodeLast = CurrentIR.at(BlockHeader->C<IROp_CodeBlock>()->Last);
|
||||
--CodeLast;
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
|
||||
// We grab these nodes this way so we can iterate easily
|
||||
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
|
||||
// Iterate the block in reverse
|
||||
while (1) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
// Optimizing flags can cause earlier flag reads to become dead but dead
|
||||
// flag reads should not impede optimiation of earlier dead flag writes.
|
||||
// We must DCE as we go to ensure we converge in a single iteration.
|
||||
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
|
||||
// Optimiation algorithm: For each flag written...
|
||||
//
|
||||
// If the flag has a later read (per FlagsRead), remove the flag from
|
||||
// FlagsRead, since the reader is covered by this write.
|
||||
//
|
||||
// Else, there is no later read, so remove the flag write (if we can).
|
||||
// This is the active part of the optimization.
|
||||
//
|
||||
// Then, add each flag read to FlagsRead.
|
||||
//
|
||||
// This order is important: instructions that read-modify-write flags
|
||||
// (like adcs) first read flags, then write flags. Since we're iterating
|
||||
// the block backwards, that means we handle the write first.
|
||||
struct FlagInfo Info = Classify(IROp);
|
||||
|
||||
if (!Info.Trivial) {
|
||||
bool Eliminated = false;
|
||||
|
||||
if ((FlagsRead & Info.Write) == 0) {
|
||||
if ((Info.CanEliminate || Info.CanReplace) && CodeNode->GetUses() == 0) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Eliminated = true;
|
||||
} else if (Info.CanReplace) {
|
||||
IROp->Op = Info.Replacement;
|
||||
}
|
||||
} else {
|
||||
FlagsRead &= ~Info.Write;
|
||||
}
|
||||
|
||||
// If we eliminated the instruction, we eliminate its read too. This
|
||||
// check is required to ensure the pass converges locally in a single
|
||||
// iteration.
|
||||
if (!Eliminated) {
|
||||
FlagsRead |= Info.Read;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Iterate in reverse
|
||||
if (CodeLast == CodeBegin) {
|
||||
break;
|
||||
}
|
||||
--CodeLast;
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->TrueBlock));
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->FalseBlock));
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(ExitOp->Args[0]));
|
||||
}
|
||||
}
|
||||
|
||||
// After processing a block, if we made progress, we must process its
|
||||
// predecessors to propagate globally. A block will be reprocessed only if
|
||||
// there is a loop backedge.
|
||||
for (; !Worklist.empty(); Worklist.pop_back()) {
|
||||
auto Block = Worklist.back();
|
||||
auto Info = CFG.Get(Block);
|
||||
Info->InWorklist = false;
|
||||
|
||||
if (ProcessBlock(IREmit, CurrentIR, Block, CFG)) {
|
||||
for (auto Pred : Info->Predecessors) {
|
||||
CFG.AddWorklist(Worklist, Pred);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fold compares into branches now that we're otherwise optimized. This needs
|
||||
// to run after eliminating carries etc and it needs the global flag metadata.
|
||||
// But it only needs to run once, we don't do it in the loop.
|
||||
for (auto [Block, _] : CurrentIR.GetBlocks()) {
|
||||
// Grab the jump
|
||||
auto BlockIROp = CurrentIR.GetOp<IR::IROp_CodeBlock>(Block);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
--CodeLast;
|
||||
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
uint32_t FlagsOut = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
|
||||
|
||||
if ((FlagsOut & FLAG_NZCV) == 0 && Op->FromNZCV) {
|
||||
FoldBranch(IREmit, CurrentIR, Op, ExitNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (CurrentIR.GetHeader()->ReadsParity) {
|
||||
OptimizeParity(IREmit, CurrentIR, CFG);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
|
||||
|
||||
@@ -28,13 +28,10 @@ namespace {
|
||||
uint32_t Available;
|
||||
uint32_t Count;
|
||||
|
||||
// If bit R of Allocated is 1, then RegToSSA[R] is the Old node
|
||||
// If bit R of Available is 0, then RegToSSA[R] is the Old node
|
||||
// currently allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to
|
||||
// clear this when freeing registers.
|
||||
Ref RegToSSA[32];
|
||||
|
||||
// Allocated base registers. Similar to ~Available except for pairs.
|
||||
uint32_t Allocated;
|
||||
};
|
||||
|
||||
IR::RegisterClassType GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
@@ -90,7 +87,7 @@ private:
|
||||
//
|
||||
// SSAToNewSSA tracks the current remapping. nullptr indicates no remapping.
|
||||
//
|
||||
// Since its indexed by Old nodes, SSAToNewSSA does not grow.
|
||||
// Since its indexed by Old nodes, SSAToNewSSA does not grow after allocation.
|
||||
fextl::vector<Ref> SSAToNewSSA;
|
||||
|
||||
// Inverse of SSAToNewSSA. Since it's indexed by new nodes, it grows.
|
||||
@@ -100,19 +97,27 @@ private:
|
||||
fextl::vector<PhysicalRegister> SSAToReg;
|
||||
|
||||
bool IsOld(Ref Node) {
|
||||
return IR->GetID(Node).Value < SSAToNewSSA.size();
|
||||
return IR->GetID(Node).Value < PreferredReg.size();
|
||||
};
|
||||
|
||||
// Return the New node (if it exists) for an Old node, else the Old node.
|
||||
Ref Map(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Pre-condition");
|
||||
|
||||
return SSAToNewSSA[IR->GetID(Old).Value] ?: Old;
|
||||
if (SSAToNewSSA.empty()) {
|
||||
return Old;
|
||||
} else {
|
||||
return SSAToNewSSA[IR->GetID(Old).Value] ?: Old;
|
||||
}
|
||||
};
|
||||
|
||||
// Return the Old node for a possibly-remapped node.
|
||||
Ref Unmap(Ref Node) {
|
||||
return NewSSAToSSA[IR->GetID(Node).Value] ?: Node;
|
||||
if (NewSSAToSSA.empty()) {
|
||||
return Node;
|
||||
} else {
|
||||
return NewSSAToSSA[IR->GetID(Node).Value] ?: Node;
|
||||
}
|
||||
};
|
||||
|
||||
// Record a remapping of Old to New.
|
||||
@@ -125,6 +130,10 @@ private:
|
||||
LOGMAN_THROW_A_FMT(NewID >= NewSSAToSSA.size(), "Brand new SSA def");
|
||||
NewSSAToSSA.resize(NewID + 1, 0);
|
||||
|
||||
if (SSAToNewSSA.empty()) {
|
||||
SSAToNewSSA.resize(PreferredReg.size(), nullptr);
|
||||
}
|
||||
|
||||
SSAToNewSSA[OldID] = New;
|
||||
NewSSAToSSA[NewID] = Old;
|
||||
|
||||
@@ -185,11 +194,11 @@ private:
|
||||
};
|
||||
|
||||
RegisterClass* GetClass(PhysicalRegister Reg) {
|
||||
return &Classes[(Reg.Class == GPRPairClass) ? GPRClass : Reg.Class];
|
||||
return &Classes[Reg.Class];
|
||||
};
|
||||
|
||||
uint32_t GetRegBits(PhysicalRegister Reg) {
|
||||
return ((Reg.Class == GPRPairClass) ? 0b11 : 0b1) << Reg.Reg;
|
||||
return 1 << Reg.Reg;
|
||||
};
|
||||
|
||||
bool IsInRegisterFile(Ref Old) {
|
||||
@@ -198,7 +207,7 @@ private:
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Allocated & GetRegBits(Reg)) && Class->RegToSSA[Reg.Reg] == Old;
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Old;
|
||||
};
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
@@ -206,10 +215,8 @@ private:
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
LOGMAN_THROW_AA_FMT((Class->Allocated & RegBits), "Register double-free");
|
||||
|
||||
Class->Available |= RegBits;
|
||||
Class->Allocated &= ~RegBits;
|
||||
};
|
||||
|
||||
bool HasSource(IROp_Header* I, Ref Old) {
|
||||
@@ -228,11 +235,14 @@ private:
|
||||
};
|
||||
|
||||
Ref DecodeSRANode(const IROp_Header* IROp, Ref Node) {
|
||||
if (IROp->Op == OP_LOADREGISTER) {
|
||||
if (IROp->Op == OP_LOADREGISTER || IROp->Op == OP_LOADPF || IROp->Op == OP_LOADAF) {
|
||||
return Node;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
return IR->GetNode(Op->Value);
|
||||
} else if (IROp->Op == OP_STOREPF || IROp->Op == OP_STOREAF) {
|
||||
const IROp_StorePF* Op = IROp->C<IR::IROp_StorePF>();
|
||||
return IR->GetNode(Op->Value);
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
@@ -242,6 +252,8 @@ private:
|
||||
RegisterClassType Class;
|
||||
uint8_t Reg;
|
||||
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
if (IROp->Op == OP_LOADREGISTER) {
|
||||
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
@@ -253,17 +265,15 @@ private:
|
||||
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Class == GPRClass || Class == FPRClass, "SRA classes");
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
if (Class == FPRClass) {
|
||||
return PhysicalRegister {FPRFixedClass, Reg};
|
||||
} else if (Reg == Core::CPUState::PF_AS_GREG) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
} else if (Reg == Core::CPUState::AF_AS_GREG) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
} else {
|
||||
return PhysicalRegister {GPRFixedClass, Reg};
|
||||
}
|
||||
@@ -273,21 +283,16 @@ private:
|
||||
// the next set bit and then clearing on each iteration.
|
||||
#define foreach_bit(b, x) for (uint32_t __x = (x), b; ((b) = __builtin_ffs(__x) - 1, __x); __x &= ~(1 << (b)))
|
||||
|
||||
void SpillReg(RegisterClass* Class, IROp_Header* Exclude, bool Pair) {
|
||||
void SpillReg(RegisterClass* Class, IROp_Header* Exclude) {
|
||||
// Find the best node to spill according to the "furthest-first" heuristic.
|
||||
// Since we defined IPs relative to the end of the block, the furthest
|
||||
// next-use has the /smallest/ unsigned IP.
|
||||
Ref Candidate = nullptr;
|
||||
uint32_t BestDistance = UINT32_MAX;
|
||||
uint8_t BestReg = ~0;
|
||||
uint32_t Allocated = ((1u << Class->Count) - 1) & ~Class->Available;
|
||||
|
||||
foreach_bit(i, Class->Allocated) {
|
||||
// We have to prioritize the pair region if we're allocating for a Pair.
|
||||
// See the comment at the call site in AssignReg.
|
||||
if (Pair && Candidate != nullptr && i >= PairRegs) {
|
||||
break;
|
||||
}
|
||||
|
||||
foreach_bit(i, Allocated) {
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Old != nullptr, "Invariant3");
|
||||
@@ -353,10 +358,8 @@ private:
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Allocated & RegBits), "Precondition");
|
||||
|
||||
Class->Available &= ~RegBits;
|
||||
Class->Allocated |= (1u << Reg.Reg);
|
||||
Class->RegToSSA[Reg.Reg] = Unmap(Node);
|
||||
|
||||
if (Index >= SSAToReg.size()) {
|
||||
@@ -366,22 +369,6 @@ private:
|
||||
SSAToReg[Index] = Reg;
|
||||
};
|
||||
|
||||
// Get the mask of available registers for a given register class
|
||||
uint32_t AvailableMask(RegisterClass* Class, bool Pair) {
|
||||
uint32_t Available = Class->Available;
|
||||
|
||||
if (Pair) {
|
||||
// Only choose base register R if R and R + 1 are both free
|
||||
Available &= (Available >> 1);
|
||||
|
||||
// Only consider aligned registers in the pair region
|
||||
constexpr uint32_t EVEN_BITS = 0x55555555;
|
||||
Available &= (EVEN_BITS & ((1u << PairRegs) - 1));
|
||||
}
|
||||
|
||||
return Available;
|
||||
};
|
||||
|
||||
// Assign a register for a given Node, spilling if necessary.
|
||||
void AssignReg(IROp_Header* IROp, Ref CodeNode, IROp_Header* Pivot) {
|
||||
const uint32_t Node = IR->GetID(CodeNode).Value;
|
||||
@@ -411,97 +398,46 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
RegisterClassType OrigClassType = GetRegClassFromNode(IR, IROp);
|
||||
bool Pair = OrigClassType == GPRPairClass;
|
||||
RegisterClassType ClassType = Pair ? GPRClass : OrigClassType;
|
||||
RegisterClass* Class = &Classes[ClassType];
|
||||
// Try to coalesce reserved pairs. Just a heuristic to remove some moves.
|
||||
if (IROp->Op == OP_ALLOCATEGPR) {
|
||||
if (IROp->C<IROp_AllocateGPR>()->ForPair) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
|
||||
// Spill to make room in the register file. Free registers need not be
|
||||
// contiguous, we'll shuffle later.
|
||||
//
|
||||
// There is one subtlety: when allocating a pair, we need at least 1 free
|
||||
// register in the pair region. Else, we could end up trying to allocate a
|
||||
// pair when the only free 2 regs are outside the pair region, and the pair
|
||||
// region is made of all pairs (so nothing to shuffle). With 1 free
|
||||
// register in the pair region, we'll be able to shuffle.
|
||||
//
|
||||
// When spilling for pairs, SpillReg prioritizes spilling the pair region
|
||||
// which ensures this loop is well-behaved.
|
||||
while (std::popcount(Class->Available) < (Pair ? 2 : 1) || (Pair && !(Class->Available & ((1u << PairRegs) - 1)))) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
SpillReg(Class, Pivot, Pair);
|
||||
}
|
||||
// Only choose base register R if R and R + 1 are both free
|
||||
Available &= (Available >> 1);
|
||||
|
||||
// There are now enough free registers, but they may be fragmented.
|
||||
// Pick a scalar blocking a pair and shuffle to make room.
|
||||
uint32_t Available = AvailableMask(Class, Pair);
|
||||
if (!Available) {
|
||||
LOGMAN_THROW_A_FMT(OrigClassType == GPRPairClass, "Already spilled");
|
||||
// Only consider aligned registers in the pair region
|
||||
constexpr uint32_t EVEN_BITS = 0x55555555;
|
||||
Available &= (EVEN_BITS & ((1u << PairRegs) - 1));
|
||||
|
||||
// Find the first free scalar. There are at least 2.
|
||||
unsigned Hole = std::countr_zero(Class->Available);
|
||||
LOGMAN_THROW_AA_FMT(Class->Available & (1u << Hole), "Definition");
|
||||
|
||||
// Its neighbour is blocking the pair.
|
||||
unsigned Blocked = Hole ^ 1;
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Available & (1u << Blocked)), "Invariant7");
|
||||
LOGMAN_THROW_AA_FMT(Hole < PairRegs, "Pairable register");
|
||||
|
||||
// Find another free scalar to evict the neighbour
|
||||
unsigned NewReg = std::countr_zero(Class->Available & ~(1u << Hole));
|
||||
LOGMAN_THROW_AA_FMT(Class->Available & (1u << NewReg), "Ensured space");
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
Ref Old = Class->RegToSSA[Blocked];
|
||||
LOGMAN_THROW_A_FMT(GetRegClassFromNode(IR, IR->GetOp<IROp_Header>(Old)) == GPRClass, "Only scalars have free neighbours");
|
||||
FreeReg(PhysicalRegister(GPRClass, Blocked));
|
||||
|
||||
Ref Clobber = nullptr;
|
||||
|
||||
// If that scalar is free because it is killed by this instruction, it
|
||||
// needs to be shuffled too, since the copy would clobber it.
|
||||
for (auto s = 0; s < IR::GetRAArgs(Pivot->Op); ++s) {
|
||||
// It is possible that the argument is to be remapped, but the actual
|
||||
// remapping in the IR only happens later in the pass so we need to
|
||||
// Map() explicitly. This can be hit with SRA shuffles.
|
||||
Ref New = Map(IR->GetNode(Pivot->Args[s]));
|
||||
const PhysicalRegister ClobberReg = SSAToReg[IR->GetID(New).Value];
|
||||
|
||||
if (ClobberReg.Class == GPRClass && ClobberReg.Reg == NewReg) {
|
||||
Clobber = IR->GetNode(Pivot->Args[s]);
|
||||
break;
|
||||
if (Available) {
|
||||
unsigned Reg = std::countr_zero(Available);
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, Reg));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Clobber) {
|
||||
// Swap the registers.
|
||||
LOGMAN_THROW_A_FMT(IsOld(Clobber), "Not yet mapped");
|
||||
|
||||
auto ClobberNew = IREmit->_Swap1(Map(Clobber), Map(Old));
|
||||
Remap(Clobber, ClobberNew);
|
||||
|
||||
auto New = IREmit->_Swap2();
|
||||
Remap(Old, New);
|
||||
|
||||
SetReg(New, PhysicalRegister(GPRClass, NewReg));
|
||||
SetReg(ClobberNew, PhysicalRegister(GPRClass, Blocked));
|
||||
FreeReg(PhysicalRegister(GPRClass, Blocked));
|
||||
} else {
|
||||
// Otherwise, simply copy.
|
||||
auto Copy = IREmit->_Copy(Map(Old));
|
||||
|
||||
Remap(Old, Copy);
|
||||
SetReg(Copy, PhysicalRegister(GPRClass, NewReg));
|
||||
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
auto After = SSAToReg[IR->GetID(IR->GetNode(IROp->Args[0])).Value];
|
||||
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
|
||||
return;
|
||||
}
|
||||
|
||||
Available = AvailableMask(Class, Pair);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Available != 0, "Post-condition of spill and shuffle");
|
||||
RegisterClassType ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClass* Class = &Classes[ClassType];
|
||||
|
||||
// Spill to make room in the register file.
|
||||
if (!Class->Available) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
SpillReg(Class, Pivot);
|
||||
}
|
||||
|
||||
// Assign a free register in the appropriate class.
|
||||
unsigned Reg = std::countr_zero(Available);
|
||||
SetReg(CodeNode, PhysicalRegister(OrigClassType, Reg));
|
||||
LOGMAN_THROW_AA_FMT(Class->Available != 0, "Post-condition of spilling");
|
||||
unsigned Reg = std::countr_zero(Class->Available);
|
||||
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
|
||||
};
|
||||
|
||||
bool IsRAOp(IROps Op) {
|
||||
@@ -530,9 +466,8 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
auto IR_ = IREmit->ViewIR();
|
||||
IR = &IR_;
|
||||
|
||||
// SSAToNewSSA, NewSSAToSSA allocated on first-use
|
||||
PreferredReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
SSAToNewSSA.resize(IR->GetSSACount(), nullptr);
|
||||
NewSSAToSSA.resize(IR->GetSSACount(), nullptr);
|
||||
SSAToReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
NextUses.resize(IR->GetSSACount(), 0);
|
||||
SpillSlotCount = 0;
|
||||
@@ -545,7 +480,6 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// At the start of each block, all registers are available.
|
||||
for (auto& Class : Classes) {
|
||||
Class.Available = (1u << Class.Count) - 1;
|
||||
Class.Allocated = 0;
|
||||
}
|
||||
|
||||
SourcesNextUses.clear();
|
||||
@@ -572,9 +506,9 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// of SourcesNextUses is consistent. The forward pass can then iterate
|
||||
// forwards and just flip the order.
|
||||
const uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
|
||||
for (int8_t i = NumArgs - 1; i >= 0; --i) {
|
||||
for (int i = NumArgs - 1; i >= 0; --i) {
|
||||
const auto& Arg = IROp->Args[i];
|
||||
if (IsValidArg(Arg)) {
|
||||
if (!Arg.IsInvalid()) {
|
||||
const uint32_t Index = Arg.ID().Value;
|
||||
|
||||
SourcesNextUses.push_back(NextUses[Index]);
|
||||
@@ -583,7 +517,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
// Record preferred registers for SRA. We also record the Node accessing
|
||||
// each register, used below. Since we initialized Class->Allocated = 0,
|
||||
// each register, used below. Since we initialized Class->Available,
|
||||
// RegToSSA is otherwise undefined so we can stash our temps there.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
@@ -636,7 +570,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
|
||||
if (Class->Allocated & (1u << Reg.Reg)) {
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
Ref Old = Class->RegToSSA[Reg.Reg];
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "RegToSSA invariant");
|
||||
@@ -684,21 +618,24 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
if (!IsValidArg(IROp->Args[s])) {
|
||||
if (IROp->Args[s].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
SourceIndex--;
|
||||
LOGMAN_THROW_AA_FMT(SourceIndex >= 0, "Consistent source count");
|
||||
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file");
|
||||
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
FreeReg(SSAToReg[IR->GetID(Map(Old)).Value]);
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
auto Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
|
||||
if (!Reg.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file");
|
||||
FreeReg(Reg);
|
||||
}
|
||||
}
|
||||
|
||||
NextUses[IR->GetID(Old).Value] = SourcesNextUses[SourceIndex];
|
||||
NextUses[IROp->Args[s].ID().Value] = SourcesNextUses[SourceIndex];
|
||||
}
|
||||
|
||||
// Assign destinations.
|
||||
@@ -707,11 +644,13 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
// Remap sources last, since AssignReg can shuffle.
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
Ref Remapped = SSAToNewSSA[IR->GetID(IR->GetNode(IROp->Args[s])).Value];
|
||||
if (!SSAToNewSSA.empty()) {
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
Ref Remapped = SSAToNewSSA[IROp->Args[s].ID().Value];
|
||||
|
||||
if (Remapped != nullptr) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, s, Remapped);
|
||||
if (Remapped != nullptr) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, s, Remapped);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Loaded 100 of 327 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user