mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 21:00:18 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cac3767c20 | ||
|
|
e82d2c70c3 | ||
|
|
9716fc73d3 | ||
|
|
aaa8eef9f1 | ||
|
|
c036868938 | ||
|
|
2726f35a10 | ||
|
|
407dc5d4dd | ||
|
|
7dda75646d | ||
|
|
b89f5b8a03 | ||
|
|
16869d8954 | ||
|
|
e353ae8408 | ||
|
|
aa3c963df1 | ||
|
|
0c8cfa7bb2 | ||
|
|
dd837fa693 | ||
|
|
eeff198ff1 | ||
|
|
0cd6371c1b | ||
|
|
5aff16f72b | ||
|
|
2c7896bce4 | ||
|
|
6dd72a09e5 | ||
|
|
4e6a4d9b69 | ||
|
|
44484a7b05 | ||
|
|
6ac7ba388f | ||
|
|
36d0a67070 | ||
|
|
0650dd1992 | ||
|
|
e2d58809ed | ||
|
|
967a74cda9 | ||
|
|
6cc6181261 | ||
|
|
f175b525f4 | ||
|
|
319f1e66cb | ||
|
|
d16969bdce | ||
|
|
c17858bfeb | ||
|
|
ec24fc3d5d | ||
|
|
8819fa88d5 | ||
|
|
5e9c2110db | ||
|
|
5aaa18e7a2 | ||
|
|
8a551b9e64 | ||
|
|
aa548bd19c | ||
|
|
9565f16d84 | ||
|
|
eb76dbdf4d | ||
|
|
967c04e252 | ||
|
|
49de1fac59 | ||
|
|
3c205eb35e | ||
|
|
682b8ef705 | ||
|
|
8c47a40625 | ||
|
|
ee5c6af868 | ||
|
|
1670c89b5e | ||
|
|
e9d34c7ded | ||
|
|
c2f2c58367 | ||
|
|
38acd9e18c | ||
|
|
21604708ca | ||
|
|
5fec6faeb6 | ||
|
|
b659701cef | ||
|
|
e902ad5278 | ||
|
|
0f54369f9b | ||
|
|
f640dcc7a4 | ||
|
|
b59da0e049 | ||
|
|
4ae3aef502 | ||
|
|
926fa3c24c | ||
|
|
8c1740f592 | ||
|
|
611aa0a5b9 | ||
|
|
0b8b5108b9 | ||
|
|
e409a0afec | ||
|
|
24211f8523 | ||
|
|
94462e4dd0 | ||
|
|
aa44668a41 | ||
|
|
33a675624c | ||
|
|
5745b419d9 | ||
|
|
9100235041 | ||
|
|
4d62e75d5a | ||
|
|
b6d3df0e54 | ||
|
|
d5d1230422 | ||
|
|
f35a212f06 | ||
|
|
f49d82deb3 | ||
|
|
77b801b8ae | ||
|
|
163a3fc899 | ||
|
|
9b627e8743 | ||
|
|
9d6865f62d | ||
|
|
d547f2b8f2 | ||
|
|
b5c9eb3463 | ||
|
|
00a3793814 | ||
|
|
4f8d7cb89c | ||
|
|
9310ac59f4 | ||
|
|
9e4de3bfe8 | ||
|
|
598b99fe58 | ||
|
|
9dda76ebe6 | ||
|
|
f5def7ae1c | ||
|
|
58bcf691d6 | ||
|
|
33186b803d | ||
|
|
bb1d7d0750 | ||
|
|
4897c1e80c | ||
|
|
e7605c94dd | ||
|
|
ab5a10374b | ||
|
|
dc2a26582f | ||
|
|
e6512fbe4d | ||
|
|
dbca441573 | ||
|
|
31adc953b7 | ||
|
|
0dd7f5e2f7 | ||
|
|
36556c4705 | ||
|
|
e29fac3f25 | ||
|
|
c984bdb42a | ||
|
|
2645b374a7 | ||
|
|
10192eebe5 | ||
|
|
0e57cbf5b9 | ||
|
|
c7413d96ad | ||
|
|
f3ba8cb33c | ||
|
|
e714933b10 | ||
|
|
9a0f83cbf6 | ||
|
|
c4d30e8437 | ||
|
|
7b30df8a56 | ||
|
|
86aa459dd1 | ||
|
|
6c9a47fc71 | ||
|
|
7a72cf631b | ||
|
|
3d98eef7b8 | ||
|
|
f2011b0b79 | ||
|
|
c0b8a0d9c3 | ||
|
|
5bd0afa307 | ||
|
|
20fb0da7d9 | ||
|
|
35cb1f710c | ||
|
|
7c7b99f9c5 | ||
|
|
09879bd962 | ||
|
|
c8f3fe3788 | ||
|
|
88e3164b55 | ||
|
|
bebd7400c0 | ||
|
|
e190d029dc | ||
|
|
37a70e2ec6 | ||
|
|
6ff073ac11 | ||
|
|
b6e4c47abc | ||
|
|
daf8b409f3 | ||
|
|
0a35029d9a | ||
|
|
e8b51ac0a4 | ||
|
|
f20e626fda | ||
|
|
2672e06aa5 | ||
|
|
4a2b4be59d | ||
|
|
05be9445cd | ||
|
|
4cb4c0e277 | ||
|
|
0135e2d78c | ||
|
|
1266a5a5ce | ||
|
|
7255850e9d | ||
|
|
002a8bebd8 | ||
|
|
452c9fd19b | ||
|
|
0b0e15c9b4 | ||
|
|
ae9db336e7 | ||
|
|
360ea538b7 | ||
|
|
f99900bdc5 | ||
|
|
34f2bce203 | ||
|
|
a05d6ae082 | ||
|
|
6a4eb434f7 | ||
|
|
0c3eb41f57 | ||
|
|
e4bfbda008 | ||
|
|
b4c797c199 | ||
|
|
ebae1bf71d | ||
|
|
fa293e5e8f | ||
|
|
720c89e648 | ||
|
|
abc5f5abd4 | ||
|
|
22ab16b217 | ||
|
|
b80f05dd86 | ||
|
|
1642a0f4d9 | ||
|
|
96e990dade | ||
|
|
3423a12ab6 | ||
|
|
c4a0fb6609 | ||
|
|
8cf23eba9c | ||
|
|
fbe817f1c4 | ||
|
|
3b61394548 | ||
|
|
8ed6b36181 | ||
|
|
9141322666 | ||
|
|
2d91c5441e | ||
|
|
2ff1589c35 | ||
|
|
a308b9edbe | ||
|
|
f6a8e595df | ||
|
|
eaaf62e4b7 | ||
|
|
dc5fc57e19 | ||
|
|
c3c0bbb060 | ||
|
|
98a91d242d | ||
|
|
8da6b1e729 | ||
|
|
d260bce31e | ||
|
|
a94aaca0ec | ||
|
|
2ac3ca3ead | ||
|
|
c559445b62 | ||
|
|
8ea4e4f663 | ||
|
|
2da700fe82 | ||
|
|
0a32ba9b29 | ||
|
|
f51832ca6b | ||
|
|
5f9a30af7a | ||
|
|
453ef0b94c | ||
|
|
4779e64cf5 | ||
|
|
e6025cc087 | ||
|
|
cd4c224dc0 | ||
|
|
07f117b957 | ||
|
|
36e4f9a070 | ||
|
|
67c751d3e5 | ||
|
|
1c59bfeb9f | ||
|
|
2b0041a291 | ||
|
|
39cc1e5116 | ||
|
|
f46856c090 | ||
|
|
b7859052da | ||
|
|
3b30d8103f | ||
|
|
d19fc36f8f | ||
|
|
5fbf76ee9e | ||
|
|
5f79761fe4 | ||
|
|
621af9e7e8 | ||
|
|
f4d107c8f0 | ||
|
|
59643db331 | ||
|
|
46f36dc89d | ||
|
|
384882744c | ||
|
|
b36d1f7e7b | ||
|
|
dd7c66db2e | ||
|
|
47ac9edd7a | ||
|
|
6e6d640fec | ||
|
|
eb88366614 | ||
|
|
50d6ddd591 | ||
|
|
249de7c758 | ||
|
|
c350888d70 | ||
|
|
f23bef1653 | ||
|
|
d17f33a922 | ||
|
|
50a3ca0d6d | ||
|
|
8745455a5b | ||
|
|
e9ab514962 | ||
|
|
4eb0948451 | ||
|
|
8d9f19bd73 | ||
|
|
9fe7bc5818 | ||
|
|
5a590a9b11 | ||
|
|
a4acd64246 | ||
|
|
304b5de1af | ||
|
|
ca2fe0b301 | ||
|
|
bbef4d762c | ||
|
|
f77841d784 | ||
|
|
219a4777c6 | ||
|
|
29f0e9b4d6 | ||
|
|
b75efc42bf | ||
|
|
90dbd4766d | ||
|
|
f7b911ca43 | ||
|
|
6de0333708 | ||
|
|
ab5d3ab22b | ||
|
|
355c3428c0 | ||
|
|
c76b7bfe8d | ||
|
|
64c5362580 | ||
|
|
77469de247 | ||
|
|
48fd827004 | ||
|
|
f09d511ac8 | ||
|
|
d5db8948db | ||
|
|
5013b8a0db | ||
|
|
ac65deed6c | ||
|
|
d1d4d2d876 | ||
|
|
e234e118e0 | ||
|
|
1cc54312fa | ||
|
|
de9ab6a023 | ||
|
|
463a0b5ed4 | ||
|
|
7b0365b377 | ||
|
|
803501526e | ||
|
|
a63a3a47e3 | ||
|
|
a66fac614b | ||
|
|
6d4693cbc1 | ||
|
|
46a2a0608b | ||
|
|
49b40b71a6 | ||
|
|
ca8f347902 | ||
|
|
e415b9443f | ||
|
|
fd4f6b8020 | ||
|
|
7a9eb01573 | ||
|
|
b0474816c7 | ||
|
|
941b065da6 | ||
|
|
ccba993ca9 | ||
|
|
e664f61da8 | ||
|
|
c74df6a59b | ||
|
|
fc76d4e8c3 | ||
|
|
099211f39c | ||
|
|
b327d7ae19 | ||
|
|
0e6a22febe | ||
|
|
b368223d50 | ||
|
|
8fe1e9562d | ||
|
|
ae545b8cf3 | ||
|
|
ac32876e4e | ||
|
|
9336e35052 | ||
|
|
0754affb98 | ||
|
|
c413d7950b | ||
|
|
f8eaf9c14f | ||
|
|
fc677eaabf | ||
|
|
114112a716 | ||
|
|
9056d9b9de | ||
|
|
74e95df661 | ||
|
|
d9544e7e02 | ||
|
|
62e1767ee0 | ||
|
|
4baeffe84f | ||
|
|
b4a67a6178 | ||
|
|
8296bfc7de | ||
|
|
f588304b12 | ||
|
|
c748dbf0e3 | ||
|
|
06497fdfad | ||
|
|
e2a7fef742 | ||
|
|
17692d6eef | ||
|
|
3020a0db2b | ||
|
|
92ddc0041b | ||
|
|
8a5388f514 | ||
|
|
a444db7dac | ||
|
|
74da2eb5fc | ||
|
|
539d5ed26f | ||
|
|
cf7ee98831 | ||
|
|
5529948479 | ||
|
|
91b6aeffe1 | ||
|
|
3a79b61f58 | ||
|
|
e92b24302f | ||
|
|
cc9ccf881b | ||
|
|
72d74ae482 | ||
|
|
e88c57bec7 | ||
|
|
ac814ac015 | ||
|
|
90f7cc925d | ||
|
|
2039950762 | ||
|
|
b526c60c73 | ||
|
|
1bfdd031d3 | ||
|
|
49ee8ba3be | ||
|
|
335cd9180e | ||
|
|
ca9a94d572 | ||
|
|
03832b2523 | ||
|
|
812224a0ef | ||
|
|
42f2851575 | ||
|
|
9c7f44f8d4 | ||
|
|
5117ba351e | ||
|
|
441fdb689d | ||
|
|
e786dfc998 | ||
|
|
a1d3183c14 | ||
|
|
c5ef0910c5 | ||
|
|
affa1d0efc | ||
|
|
a83dc27a42 | ||
|
|
129ec63e92 | ||
|
|
3459369c6e | ||
|
|
7a490a3811 | ||
|
|
ee17fe239a | ||
|
|
0416950aaa | ||
|
|
6e46383cdb | ||
|
|
faf1b85904 | ||
|
|
bec5f4fe2a | ||
|
|
fe6dbbfa63 | ||
|
|
b19440c78f | ||
|
|
abf9700475 | ||
|
|
8d6b454455 | ||
|
|
205ec3e14d | ||
|
|
f897579593 | ||
|
|
2478abba29 | ||
|
|
a4db585664 | ||
|
|
8bf4a124c8 | ||
|
|
13c3b65732 | ||
|
|
7e6ba184f9 | ||
|
|
f6cb914a2a | ||
|
|
6c92f94ee8 | ||
|
|
bebf4209d6 | ||
|
|
c82a683987 | ||
|
|
886d40ccfe | ||
|
|
8f33e56e21 | ||
|
|
b1634680fe | ||
|
|
54d332935e | ||
|
|
2829ad56a1 | ||
|
|
fbf62f1296 | ||
|
|
e6aa268093 | ||
|
|
cc589ba7e6 | ||
|
|
f009a00986 | ||
|
|
8617150a42 | ||
|
|
ef823ce82b | ||
|
|
fc2917117e | ||
|
|
68abe400a7 | ||
|
|
df86d80a85 | ||
|
|
58cff72d38 | ||
|
|
fd1f5643c7 | ||
|
|
0ea29dfbdf | ||
|
|
f4fca4482f | ||
|
|
86d1ce7f00 | ||
|
|
15fef1794c | ||
|
|
8b5d9d8fdf | ||
|
|
0ec724cf1f | ||
|
|
0dc0117e86 | ||
|
|
1d00ad6030 | ||
|
|
23a076c313 | ||
|
|
57eacab654 | ||
|
|
b4093a8888 | ||
|
|
7c5a9b5d6a | ||
|
|
12b3c82d83 | ||
|
|
66520bce0a | ||
|
|
25cc2bdcb8 | ||
|
|
f98b18800c | ||
|
|
0dd687a7a1 | ||
|
|
1aff3acbb9 | ||
|
|
a6ab2ca30d | ||
|
|
ca43e2a61c | ||
|
|
4f404160d0 | ||
|
|
5edc69b692 | ||
|
|
b6cb897896 | ||
|
|
47b3637452 | ||
|
|
5b65f30c8f | ||
|
|
23572539f8 | ||
|
|
7176c717e5 | ||
|
|
3a05b760d4 | ||
|
|
1e0273d570 | ||
|
|
8232be6302 | ||
|
|
43cd897cf4 | ||
|
|
e593807856 | ||
|
|
ce8e6e5c0c | ||
|
|
5b9a7c7845 | ||
|
|
0e5f5a2db9 | ||
|
|
50a9cea16e | ||
|
|
6aa95d82f2 | ||
|
|
7c34f449d1 | ||
|
|
e9435203a0 | ||
|
|
8d9ba0e102 | ||
|
|
3fcfd2ccde | ||
|
|
aa4205c2e8 | ||
|
|
5d5b411611 | ||
|
|
5f5a06fb3b | ||
|
|
3077addcf8 | ||
|
|
8b6fe0c4ff | ||
|
|
29fd62e9ba | ||
|
|
2631b113da | ||
|
|
da152031b3 | ||
|
|
be9c5678c0 | ||
|
|
fdd370dd1a | ||
|
|
6951284924 | ||
|
|
b03c613fbf | ||
|
|
b893bdb8df | ||
|
|
db45f6eec8 | ||
|
|
d53e689e22 | ||
|
|
f55378257a | ||
|
|
6a7914ac56 | ||
|
|
d935d25f0c | ||
|
|
84cf1d2fd2 | ||
|
|
ed0c045c17 | ||
|
|
d585063e60 | ||
|
|
3077fec9ab | ||
|
|
9500842efc | ||
|
|
a6f9c51317 | ||
|
|
cfa2ad8423 | ||
|
|
19010491da | ||
|
|
5d613e8716 | ||
|
|
894aaa980f | ||
|
|
877b2f4fef | ||
|
|
2a170cfdec | ||
|
|
a299d6b1a5 | ||
|
|
ffb85e6305 | ||
|
|
d7a20fa28f | ||
|
|
5ac7d5dfcd | ||
|
|
75644b33df | ||
|
|
4c4c6e7807 | ||
|
|
a8c9c71ce3 | ||
|
|
8a4bd5f22c | ||
|
|
138a36c69a | ||
|
|
3400ca5d42 | ||
|
|
5d8164da4e | ||
|
|
8b89b30a6f | ||
|
|
e66d7cfd6c | ||
|
|
63afa29dae | ||
|
|
926a9b40e2 | ||
|
|
9bb43264b5 | ||
|
|
32d6daf558 | ||
|
|
bebcb73c68 | ||
|
|
d0e040514f | ||
|
|
850c027d52 | ||
|
|
5df90563e2 | ||
|
|
7a489d18c6 | ||
|
|
1db092e96f | ||
|
|
65a4de221b | ||
|
|
9c8df79dfb | ||
|
|
689b461d7b | ||
|
|
92c951c81f | ||
|
|
9c8438f264 | ||
|
|
caf7ad53e6 | ||
|
|
47f0fec2f2 | ||
|
|
eadb502059 | ||
|
|
8e1695afa3 | ||
|
|
a7138f26b6 | ||
|
|
f2a9ce9d4c | ||
|
|
84e4960e52 | ||
|
|
99afd876ba | ||
|
|
1caa31c5cb | ||
|
|
d2c82ba707 | ||
|
|
9067f3513a | ||
|
|
409d691877 | ||
|
|
1762ff0393 | ||
|
|
4d26178f25 | ||
|
|
c6582a1ce5 | ||
|
|
a51be56c9e | ||
|
|
e5cb583cc0 | ||
|
|
3ef83a9e23 | ||
|
|
7de6a5cb07 | ||
|
|
e06b3a4186 | ||
|
|
c495f82f4f | ||
|
|
49e4426ed5 | ||
|
|
c4e2436885 | ||
|
|
f078b25c7d | ||
|
|
e5149fba57 | ||
|
|
96055cbde7 | ||
|
|
4abac0cac7 | ||
|
|
1e1bcc4af2 | ||
|
|
f1d7879365 | ||
|
|
0ea3de95e0 | ||
|
|
33cef7c8cd | ||
|
|
b0fd220f3e | ||
|
|
1e3539537d | ||
|
|
1b13a3dd4e | ||
|
|
d1ec242e4f | ||
|
|
21f9841e9d | ||
|
|
ea6c05a46a | ||
|
|
226f5e2f23 | ||
|
|
a82fcdecd7 | ||
|
|
27acbe305d | ||
|
|
cd0739a534 | ||
|
|
f1055d0713 | ||
|
|
7875b20594 | ||
|
|
25679cd319 | ||
|
|
fa45ec32db | ||
|
|
40d12c4d0e | ||
|
|
5e706dff64 | ||
|
|
a4860d6f87 | ||
|
|
ef4c4f6e9b | ||
|
|
abbd65549a | ||
|
|
00ef1baead | ||
|
|
75687070a5 | ||
|
|
9b1496c327 | ||
|
|
27cce9d9b7 | ||
|
|
08b66bf827 | ||
|
|
0aa5807482 | ||
|
|
7a2d8c5c01 | ||
|
|
1b8b1a24b0 | ||
|
|
86eb2b13b9 | ||
|
|
c634c53434 | ||
|
|
2eb7a9ff28 | ||
|
|
933c65d805 | ||
|
|
df0ecad15b | ||
|
|
ce88f5f948 | ||
|
|
27cd399d6f | ||
|
|
7b70925acf | ||
|
|
2a3e0b93e8 | ||
|
|
2b26d0fff5 | ||
|
|
129f676610 | ||
|
|
2dc92c122e | ||
|
|
0db17bd58a | ||
|
|
aac16493f1 | ||
|
|
86e5e1a15e | ||
|
|
aa5d2ff31c | ||
|
|
30dd5f5c75 | ||
|
|
f5bc06476a | ||
|
|
c70e44cf05 | ||
|
|
ede13e37d9 | ||
|
|
a7dada457a | ||
|
|
6367554d30 | ||
|
|
81969a684e | ||
|
|
ee339b5960 | ||
|
|
881c940693 | ||
|
|
900c62fa7b | ||
|
|
200c6c054f | ||
|
|
bc0927b7b1 | ||
|
|
231c28395c | ||
|
|
64a45c0d29 | ||
|
|
cab02be637 | ||
|
|
74f341bc0e | ||
|
|
feaa1af1a8 | ||
|
|
d59d040b4e | ||
|
|
a9c26cbf71 | ||
|
|
f4b5c4e69a | ||
|
|
b8cac9f7d5 | ||
|
|
13974df204 | ||
|
|
fa6fe9bf06 | ||
|
|
746be0824e | ||
|
|
93120cabbb | ||
|
|
04ae05f4ce | ||
|
|
83773dddc7 | ||
|
|
9813553f02 | ||
|
|
924723d433 | ||
|
|
33558e63c4 | ||
|
|
97c229d5eb | ||
|
|
16b007df33 | ||
|
|
3cbc421c7e | ||
|
|
890e5e1f0f | ||
|
|
6b98454f03 | ||
|
|
b05329c8df | ||
|
|
6f43c8ffac | ||
|
|
ffac98051f | ||
|
|
40812efaae | ||
|
|
3429321d59 | ||
|
|
6cddd6cbe7 | ||
|
|
23d07d7d0c | ||
|
|
1351575713 | ||
|
|
8aa7d1a278 | ||
|
|
fb60a8a032 | ||
|
|
f40bc134df | ||
|
|
f3811f04bd | ||
|
|
94bb7eb311 | ||
|
|
3383786205 | ||
|
|
91f4c54768 | ||
|
|
d9c779289c | ||
|
|
5631ff4fd5 | ||
|
|
34301319bf | ||
|
|
8eac3198b6 | ||
|
|
832edd4da3 | ||
|
|
5823e74bcd | ||
|
|
a4545f493e | ||
|
|
3b8cd44ca4 | ||
|
|
f138d7d9b8 | ||
|
|
3bb9d44bf5 | ||
|
|
06e6a1b19e | ||
|
|
256d166126 | ||
|
|
630285c589 | ||
|
|
9621eca677 | ||
|
|
7eee50d929 | ||
|
|
1dc22e33ae | ||
|
|
0ef8aaebeb | ||
|
|
d17c427e47 | ||
|
|
f73fb62c6e | ||
|
|
5976b712ca | ||
|
|
63f5e64adb | ||
|
|
34ee3bb8aa | ||
|
|
79c745929a | ||
|
|
8330cc6876 | ||
|
|
03ca3e7e68 | ||
|
|
2c3e6cbe65 | ||
|
|
16e6163677 | ||
|
|
bbbc0dc9dc | ||
|
|
f9bdf0bd01 | ||
|
|
4afc7adb05 | ||
|
|
2152d1b2e9 | ||
|
|
0ecfc651b6 | ||
|
|
4a3250ddea | ||
|
|
633f624a69 | ||
|
|
ea4004ec9d | ||
|
|
a469047e7a | ||
|
|
26a8a2717c | ||
|
|
4c0e6d5779 | ||
|
|
a350ef5d1b | ||
|
|
2cc0a867a9 | ||
|
|
9d9bd750e2 | ||
|
|
85d1b573ef | ||
|
|
2f8c5b4820 | ||
|
|
a1f55f0b0b | ||
|
|
7b1d9540b7 | ||
|
|
1007f874bf | ||
|
|
24ea4b7537 | ||
|
|
93ab57454a | ||
|
|
4882f10536 | ||
|
|
fe43a2bcb2 | ||
|
|
6700511cdf | ||
|
|
e836639427 | ||
|
|
7eb3f33162 | ||
|
|
b35a514ea6 | ||
|
|
8f372821f8 | ||
|
|
6376d06bed | ||
|
|
1bb293d4be | ||
|
|
0e069f1e97 | ||
|
|
9074b810a9 | ||
|
|
c5fe8723d3 | ||
|
|
434bffac33 | ||
|
|
e84848b16b | ||
|
|
69ed39d49e | ||
|
|
230bde6aef | ||
|
|
c24d7aacba | ||
|
|
e613876e9d | ||
|
|
1473129a8f | ||
|
|
c42808cb70 | ||
|
|
054c119e2e | ||
|
|
f75bd2f09b | ||
|
|
a7424416d9 | ||
|
|
cadb0a2ddb | ||
|
|
2da819c0f3 | ||
|
|
e0c783de74 | ||
|
|
6c003fcb9a | ||
|
|
c4faffc0e2 | ||
|
|
cc2d21f411 | ||
|
|
59686a6c60 | ||
|
|
0ab864da17 | ||
|
|
3c32271dd0 | ||
|
|
507a95b817 | ||
|
|
9bb9e954c2 | ||
|
|
7f3582bb23 | ||
|
|
efe15ce336 | ||
|
|
21b0f35ef4 | ||
|
|
ccf332d48e | ||
|
|
802a32ce8a | ||
|
|
70c02d5c58 | ||
|
|
2e4fb47848 | ||
|
|
a4d5302369 | ||
|
|
6ff3c90af3 | ||
|
|
c114279118 | ||
|
|
7ffd3e55d5 | ||
|
|
201fe6ee23 | ||
|
|
83fedd6c8f | ||
|
|
dedf4a93d0 | ||
|
|
1f59f0e226 | ||
|
|
c3c2b6115d | ||
|
|
115fbb5039 | ||
|
|
27973d5637 | ||
|
|
f121be649d | ||
|
|
af9bcb3efd | ||
|
|
4877bb3f19 | ||
|
|
e2583249c6 | ||
|
|
f48071dcbe | ||
|
|
fa72be2ec5 | ||
|
|
b7ff6dd9c6 | ||
|
|
cb6d60aa87 | ||
|
|
80c9a43eef | ||
|
|
05155778d4 | ||
|
|
49b8dae189 | ||
|
|
3e59fc0a8c | ||
|
|
f98c010854 | ||
|
|
10ee963f52 | ||
|
|
af7462ee6a | ||
|
|
be4777110c | ||
|
|
696503680a | ||
|
|
dd4d3bcf38 | ||
|
|
f26bb6bf53 | ||
|
|
2c4fd79304 | ||
|
|
910ec4aadd | ||
|
|
fb7275b3d8 | ||
|
|
51b4bfc6a6 | ||
|
|
9f7bc94f9d | ||
|
|
069e2ce62a | ||
|
|
5bbbced1bd | ||
|
|
941fd9c6ea | ||
|
|
3332220d06 | ||
|
|
5933a59c09 | ||
|
|
aee8c9def2 | ||
|
|
c2136272bf | ||
|
|
dd26b0c879 | ||
|
|
3d9114b74b | ||
|
|
3f3e937967 | ||
|
|
4beb29141b | ||
|
|
9a8e7eaace | ||
|
|
4227e012aa | ||
|
|
40bbcb7061 | ||
|
|
67663b812e | ||
|
|
d24d0a95a0 | ||
|
|
87fbcf754d | ||
|
|
403fd62b34 | ||
|
|
d92b6a9ac4 | ||
|
|
c2092bfed0 | ||
|
|
a6cf7fa508 | ||
|
|
5ff09f5091 | ||
|
|
9af7ee6bd2 | ||
|
|
d1e36f264f | ||
|
|
b1ec50c7c2 | ||
|
|
93eead243f | ||
|
|
ceac38a6ac | ||
|
|
3b2e657fd4 | ||
|
|
380ba0a014 | ||
|
|
1fe497d1dd | ||
|
|
7816b150d0 | ||
|
|
ce8bc9d25c | ||
|
|
4634688aca | ||
|
|
dd3e3ed189 | ||
|
|
f8ef6feff9 | ||
|
|
9201ac5a6b | ||
|
|
8ebf049fb9 | ||
|
|
3c5b59d985 | ||
|
|
6b91e0cb0e | ||
|
|
587b924de9 | ||
|
|
d507f4c9b1 | ||
|
|
39bc2a82c1 | ||
|
|
774325dcf2 | ||
|
|
a1378f94ce | ||
|
|
77ec950ff2 | ||
|
|
592d6cc43f | ||
|
|
610caf8529 | ||
|
|
d20b46e46f | ||
|
|
4094aa1b9a | ||
|
|
f8c6baae97 | ||
|
|
5c9bb6594c | ||
|
|
56df57e980 | ||
|
|
f7b4d25803 | ||
|
|
4fffe68f81 | ||
|
|
95b15d788b | ||
|
|
ae9312bdab | ||
|
|
b78da2e5ad | ||
|
|
54fc8cb0bd | ||
|
|
228009c283 | ||
|
|
1f878ce4cd | ||
|
|
e4b7a65a49 | ||
|
|
c77a707dbe | ||
|
|
f81fc4e4f0 | ||
|
|
d385e496d3 | ||
|
|
f8c4c543e3 | ||
|
|
d2f903ae55 | ||
|
|
d1249ec5cf | ||
|
|
bf9a6d763c | ||
|
|
0b829d2c46 | ||
|
|
bddb533fa0 | ||
|
|
1c35eeffeb | ||
|
|
c7254e31ed | ||
|
|
b0bd8a62a2 | ||
|
|
17a55fbb39 | ||
|
|
dedec83881 | ||
|
|
9fd5c73633 | ||
|
|
bdb890a8b0 | ||
|
|
5043d09771 | ||
|
|
da51169ba9 | ||
|
|
f72cee480f | ||
|
|
7546160811 | ||
|
|
e7d5a01c5f | ||
|
|
0c3a8d0bc8 | ||
|
|
19e58cac62 | ||
|
|
c4ba7eee87 | ||
|
|
09c4a5594a | ||
|
|
d204155661 | ||
|
|
924b8c10a9 | ||
|
|
9017cd14c8 | ||
|
|
8d89adef2e | ||
|
|
6615b55c12 | ||
|
|
66865dd177 | ||
|
|
1e709d1150 | ||
|
|
476ee0cd7d | ||
|
|
6df51a57b3 | ||
|
|
e65545a537 | ||
|
|
b8e864ffdf | ||
|
|
ed87c01470 | ||
|
|
6c21b86a8f | ||
|
|
d79b7fcc49 | ||
|
|
b9a6caea8d | ||
|
|
9688f5e17d | ||
|
|
f6f8d26426 | ||
|
|
dba0a1d09e | ||
|
|
af3145674e | ||
|
|
3c19e634b3 | ||
|
|
f964a5187e | ||
|
|
8e0fdfc325 | ||
|
|
839f9ecd3b | ||
|
|
95fc69b628 | ||
|
|
b9da95838a | ||
|
|
1059279d5d | ||
|
|
5dc85307a6 | ||
|
|
549e06aade | ||
|
|
3b189f6d7d | ||
|
|
a5d3692b53 | ||
|
|
97a68cb643 | ||
|
|
d1b5dfd4b1 | ||
|
|
6cdaea680d | ||
|
|
870e395ac4 | ||
|
|
04592f82f5 | ||
|
|
19e849283f | ||
|
|
b6e1469cd4 | ||
|
|
5ef0db994d | ||
|
|
b523407a3e | ||
|
|
8021dc10a1 | ||
|
|
7e8d734e43 | ||
|
|
3d90d1ab4f | ||
|
|
3c7318d7c8 | ||
|
|
fc0b233046 | ||
|
|
e25918d846 | ||
|
|
d78b0ea435 | ||
|
|
3a334c4585 | ||
|
|
8dae4bcd44 | ||
|
|
294f10fdd0 | ||
|
|
b9829ed316 | ||
|
|
4ccec17676 | ||
|
|
f45082043b | ||
|
|
3222f13dde | ||
|
|
b282620a48 | ||
|
|
3ff1ff8f74 | ||
|
|
e24b01b6cb | ||
|
|
9a8694c2f3 | ||
|
|
070a9148aa | ||
|
|
f19fe3b6f3 | ||
|
|
8d2b15665d | ||
|
|
4dec8f22f8 | ||
|
|
a39b3aca78 | ||
|
|
5dc4ab062d | ||
|
|
31f82c1d96 | ||
|
|
548fd9daf8 | ||
|
|
f831f5a0e1 | ||
|
|
4c21aa2604 | ||
|
|
c9efb75714 | ||
|
|
5e56bdc0fd | ||
|
|
3554d5c2f7 | ||
|
|
5fe405e1fb | ||
|
|
8381d44bbd | ||
|
|
56bb3744a5 | ||
|
|
470b435afd | ||
|
|
a4f8bbff02 | ||
|
|
cf5ab05b90 | ||
|
|
3a2ce240f9 | ||
|
|
1f01dd53f7 | ||
|
|
72d41d70b6 | ||
|
|
42b5b1f64c | ||
|
|
2949bc211d | ||
|
|
5e0952159d | ||
|
|
441187470e | ||
|
|
59fd13cc2f | ||
|
|
c9e7bfdf16 | ||
|
|
2d700c381e | ||
|
|
72d6c8ebd6 | ||
|
|
991c6941c1 | ||
|
|
f974696e34 | ||
|
|
3ef9ea94e5 | ||
|
|
381ce23fd7 | ||
|
|
af6a0be832 | ||
|
|
287fe5beac | ||
|
|
b9c214e6e8 | ||
|
|
d3d76aa8ce | ||
|
|
3bea08da5f | ||
|
|
3d5cacbdc3 | ||
|
|
7ccb252069 | ||
|
|
31547462bb | ||
|
|
b3a7a973a1 | ||
|
|
22b26696ba | ||
|
|
495241f8ca | ||
|
|
4afbfcae17 | ||
|
|
3627de4cbc | ||
|
|
007c07e612 | ||
|
|
ec7c8fd922 | ||
|
|
c5a0ae7b34 | ||
|
|
4bd207ebf3 | ||
|
|
45011234d9 | ||
|
|
24017f379e | ||
|
|
aad7656b38 | ||
|
|
80de890f05 | ||
|
|
62cec7b6b2 | ||
|
|
c9c163cd7b | ||
|
|
95a9f32bf0 | ||
|
|
c4ae761a0e | ||
|
|
0653b346e0 | ||
|
|
fa587398bd | ||
|
|
6b67857151 | ||
|
|
81165f0c40 | ||
|
|
df40515087 | ||
|
|
c77922e3e5 | ||
|
|
0f9abe68b9 | ||
|
|
c168ee6940 | ||
|
|
0d4414fdd0 | ||
|
|
968d5e0d8f | ||
|
|
635182b57c | ||
|
|
9d0b6ce75e | ||
|
|
2fdd80fe3a | ||
|
|
dbac23b749 | ||
|
|
7fa7061aa5 | ||
|
|
e45e631199 | ||
|
|
b21e77c1e0 | ||
|
|
ba33294225 | ||
|
|
97c21cc3a7 | ||
|
|
7d7e6f5326 | ||
|
|
5e15bd935e | ||
|
|
9bad09c45f | ||
|
|
47d077ff22 | ||
|
|
bbf8dde3ca | ||
|
|
6e8ca3bc6c | ||
|
|
11a494d7b3 | ||
|
|
9b570de33f | ||
|
|
b67343fc5a | ||
|
|
5a3c0eb83c | ||
|
|
10391608a0 | ||
|
|
51c57cc5ae | ||
|
|
05e4678e65 | ||
|
|
0f0e402db4 | ||
|
|
837bccb1d8 | ||
|
|
adc709db2f | ||
|
|
395573720d | ||
|
|
0e62759d24 | ||
|
|
926b6c3117 | ||
|
|
c9f9304ba5 | ||
|
|
fabd6be5af | ||
|
|
1bf31d20b6 | ||
|
|
653bf04db0 | ||
|
|
b77a25b21a | ||
|
|
9db6931cea | ||
|
|
bad5cef52b | ||
|
|
94bd79b2bf | ||
|
|
b746146f4e | ||
|
|
8ac9bb5c72 | ||
|
|
1b552a6f62 | ||
|
|
97329ccc7a | ||
|
|
f2d1f2de56 | ||
|
|
692c2fae96 | ||
|
|
3d65b701a2 | ||
|
|
ecaca0fe15 | ||
|
|
8955f83ef6 | ||
|
|
1a8aaebd79 | ||
|
|
38a823cc54 | ||
|
|
25306cb373 | ||
|
|
1084a031e7 | ||
|
|
f6ec99bede | ||
|
|
90a6647fa4 | ||
|
|
a926bb81a9 | ||
|
|
fbf41e3149 | ||
|
|
a38205069b | ||
|
|
2d75801024 | ||
|
|
504511fe7e |
No files matched your search
@@ -16,7 +16,7 @@ jobs:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, ARM64, mingw]]
|
||||
arch: [[self-hosted, ARM64, mingw], [self-hosted, ARM64EC, mingw, ARM64]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
@@ -38,6 +38,11 @@ jobs:
|
||||
run: |
|
||||
echo "MINGW_TRIPLE=aarch64-w64-mingw32" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC Arm64EC
|
||||
if: matrix.arch[1] == 'ARM64EC'
|
||||
run: |
|
||||
echo "MINGW_TRIPLE=arm64ec-w64-mingw32" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
@@ -73,7 +78,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -72,23 +72,22 @@ jobs:
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: ASM Tests
|
||||
- name: ASM Tests - SVE256
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
- name: ASM Test SVE256 Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_SVE256Bit.log || true
|
||||
|
||||
- name: ASM Tests 128-bit
|
||||
- name: ASM Tests - SVE128
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
FEX_HOSTFEATURES: "disableavx"
|
||||
FEX_FORCESVEWIDTH: "128"
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
@@ -97,7 +96,21 @@ jobs:
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM128bit.log || true
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_SVE128Bit.log || true
|
||||
|
||||
- name: ASM Tests - ASIMD
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
FEX_HOSTFEATURES: "disablesve"
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test ASIMD Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_ASIMD.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@ compile_commands.json
|
||||
vim_rc
|
||||
Config.json
|
||||
|
||||
[Bb]uild*/
|
||||
[Bb]uild*
|
||||
[Bb]in/
|
||||
out/
|
||||
.vscode/
|
||||
|
||||
@@ -5,15 +5,6 @@
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/imgui"]
|
||||
path = External/imgui
|
||||
url = https://github.com/Sonicadvance1/imgui.git
|
||||
[submodule "External/json-maker"]
|
||||
path = External/json-maker
|
||||
url = https://github.com/Sonicadvance1/json-maker.git
|
||||
[submodule "External/tiny-json"]
|
||||
path = External/tiny-json
|
||||
url = https://github.com/Sonicadvance1/tiny-json.git
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
|
||||
+83
-34
@@ -1,5 +1,5 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX)
|
||||
project(FEX C CXX ASM)
|
||||
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
@@ -7,7 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -15,6 +15,7 @@ option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_COVERAGE "Enables Coverage" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
@@ -27,10 +28,12 @@ option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
@@ -40,12 +43,13 @@ string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
message (STATUS "Mingw build")
|
||||
set (MINGW_BUILD TRUE)
|
||||
set (ENABLE_JEMALLOC FALSE)
|
||||
set (ENABLE_JEMALLOC TRUE)
|
||||
set (ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
set (CLANG_MINIMUM_VERSION 12.0)
|
||||
set (CLANG_MINIMUM_VERSION 13.0)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
|
||||
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
endif()
|
||||
@@ -137,9 +141,39 @@ endif()
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
|
||||
set(_M_ARM_64EC 1)
|
||||
add_definitions(-D_M_ARM_64EC=1)
|
||||
endif()
|
||||
|
||||
# Required as FEX is not allowed to lock the CRT heap lock during compilation or callbacks
|
||||
set(ENABLE_JEMALLOC TRUE)
|
||||
include(CheckCXXSourceCompiles)
|
||||
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
__attribute__((preserve_all))
|
||||
int Testy(int a, int b, int c, int d, int e, int f) {
|
||||
return a + b + c + d + e + f;
|
||||
}
|
||||
int main() {
|
||||
return Testy(0, 1, 2, 3, 4, 5);
|
||||
}"
|
||||
HAS_CLANG_PRESERVE_ALL)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
if (HAS_CLANG_PRESERVE_ALL)
|
||||
if (MINGW_BUILD)
|
||||
message(STATUS "Ignoring broken clang::preserve_all support")
|
||||
set(HAS_CLANG_PRESERVE_ALL FALSE)
|
||||
else()
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
add_definitions("-DFEX_PRESERVE_ALL_ATTR=__attribute__((preserve_all))" "-DFEX_HAS_PRESERVE_ALL_ATTR=1")
|
||||
else()
|
||||
add_definitions("-DFEX_PRESERVE_ALL_ATTR=" "-DFEX_HAS_PRESERVE_ALL_ATTR=0")
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# We can run the simulator on both x86-64 or AArch64 hosts
|
||||
add_definitions(-DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_CCACHE)
|
||||
@@ -189,13 +223,17 @@ if (ENABLE_TSAN)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
endif()
|
||||
|
||||
if (ENABLE_COVERAGE)
|
||||
add_compile_options(-fprofile-instr-generate -fcoverage-mapping)
|
||||
link_libraries(-fprofile-instr-generate -fcoverage-mapping)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
# The glibc jemalloc subproject which hooks the glibc allocator.
|
||||
# Required for thunks to work.
|
||||
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
|
||||
add_definitions(-DENABLE_JEMALLOC_GLIBC=1)
|
||||
add_subdirectory(External/jemalloc_glibc/)
|
||||
else()
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
" jemalloc glibc allocator disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
@@ -205,10 +243,8 @@ endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
# The jemalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
else()
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
" jemalloc disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
@@ -216,6 +252,11 @@ else()
|
||||
" Use at your own risk!")
|
||||
endif()
|
||||
|
||||
if (USE_PDB_DEBUGINFO)
|
||||
add_compile_options(-g -gcodeview)
|
||||
add_link_options(-g -Wl,--pdb=)
|
||||
endif()
|
||||
|
||||
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
|
||||
@@ -229,8 +270,10 @@ if (BUILD_TESTS)
|
||||
set(COMPILE_VIXL_DISASSEMBLER TRUE)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
if (COMPILE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
@@ -240,37 +283,42 @@ endif()
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
|
||||
pkg_search_module(xxhash IMPORTED_TARGET xxhash libxxhash)
|
||||
if (TARGET PkgConfig::xxhash AND NOT CMAKE_CROSSCOMPILING)
|
||||
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
|
||||
else()
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
endif()
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(External/Catch2/)
|
||||
find_package(Catch2 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
endif()
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
|
||||
add_subdirectory(External/json-maker/)
|
||||
include_directories(External/json-maker/)
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
include_directories(External/tiny-json/)
|
||||
|
||||
include_directories(External/xbyak/)
|
||||
|
||||
include_directories(Source/)
|
||||
include_directories("${CMAKE_BINARY_DIR}/Source/")
|
||||
|
||||
@@ -372,8 +420,7 @@ if (BUILD_TESTS)
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
message(STATUS "Running tests with ${TEST_JOB_COUNT} jobs")
|
||||
endif()
|
||||
if (CMAKE_VERSION VERSION_LESS "3.29")
|
||||
elseif(CMAKE_VERSION VERSION_LESS "3.29")
|
||||
execute_process(COMMAND "nproc" OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE TEST_JOB_COUNT)
|
||||
endif()
|
||||
set(TEST_JOB_FLAG "-j${TEST_JOB_COUNT}")
|
||||
@@ -383,8 +430,10 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
# This is a reference AArch64 cross compile script
|
||||
# Pass in to cmake when building:
|
||||
# eg: cmake -DCMAKE_TOOLCHAIN_FILE=../CMakeToolchains/AArch64.cmake ..
|
||||
if (NOT DEFINED ENV{SYSROOT})
|
||||
message(FATAL_ERROR "Need to have SYSROOT environment variable set")
|
||||
endif()
|
||||
|
||||
set(CMAKE_SYSTEM_NAME Linux)
|
||||
set(CMAKE_SYSTEM_PROCESSOR aarch64)
|
||||
set(CMAKE_CROSSCOMPILING TRUE)
|
||||
|
||||
# Target triple needs to match the binutils exactly
|
||||
set(TARGET_TRIPLE aarch64-linux-gnu)
|
||||
set(CMAKE_C_COMPILER "clang")
|
||||
set(CMAKE_CXX_COMPILER "clang++")
|
||||
set(CMAKE_C_COMPILER_AR "llvm-ar")
|
||||
set(CMAKE_CXX_COMPILER_AR "llvm-ar")
|
||||
set(CMAKE_C_COMPILER_RANLIB "llvm-ranlib")
|
||||
set(CMAKE_CXX_COMPILER_RANLIB "llvm-ranlib")
|
||||
set(CMAKE_LINKER "ld.lld")
|
||||
|
||||
set(CMAKE_C_COMPILER_TARGET ${TARGET_TRIPLE})
|
||||
set(CMAKE_CXX_COMPILER_TARGET ${TARGET_TRIPLE})
|
||||
|
||||
# Set the environment variable SYSROOT to the aarch64 rootfs
|
||||
set(CMAKE_FIND_ROOT_PATH "$ENV{SYSROOT}")
|
||||
set(CMAKE_SYSROOT "$ENV{SYSROOT}")
|
||||
|
||||
list(APPEND CMAKE_PREFIX_PATH "$ENV{SYSROOT}/usr/lib/${TARGET_TRIPLE}/cmake/")
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER)
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY)
|
||||
@@ -301,8 +301,9 @@ public:
|
||||
xbfiz_helper(true, s, rd, rn, lsb, width);
|
||||
}
|
||||
void asr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
sbfm(s, rd, rn, shift, RegSize_m1);
|
||||
}
|
||||
|
||||
void uxtb(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
@@ -325,14 +326,14 @@ public:
|
||||
}
|
||||
|
||||
void lsl(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
ubfm(s, rd, rn, (RegSize - shift) % RegSize, RegSize - shift - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
ubfm(s, rd, rn, (RegSizeInBits(s) - shift) & RegSize_m1, RegSize_m1 - shift);
|
||||
}
|
||||
void lsr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsr a region larger than the register");
|
||||
ubfm(s, rd, rn, shift, RegSize - 1);
|
||||
const auto RegSize_m1 = RegSizeInBits(s) - 1;
|
||||
shift &= RegSize_m1;
|
||||
ubfm(s, rd, rn, shift, RegSize_m1);
|
||||
}
|
||||
void ubfx(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
LOGMAN_THROW_A_FMT(width > 0, "ubfx needs width > 0");
|
||||
@@ -368,6 +369,7 @@ public:
|
||||
}
|
||||
|
||||
void ror(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t Imm) {
|
||||
Imm &= RegSizeInBits(s) - 1;
|
||||
extr(s, rd, rn, rn, Imm);
|
||||
}
|
||||
|
||||
|
||||
@@ -1070,7 +1070,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10110, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1082,7 +1082,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10110, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1095,7 +1095,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10111, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1107,7 +1107,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 0, ConvertedSize, 0b10111, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1123,7 +1123,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11000, rd, rn);
|
||||
}
|
||||
@@ -1138,7 +1138,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11001, rd, rn);
|
||||
}
|
||||
@@ -1154,7 +1154,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
@@ -1169,7 +1169,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
@@ -1184,7 +1184,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
@@ -1199,7 +1199,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -1214,7 +1214,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11110, rd, rn);
|
||||
}
|
||||
@@ -1229,7 +1229,7 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize =
|
||||
size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b11111, rd, rn);
|
||||
}
|
||||
|
||||
@@ -602,6 +602,18 @@ constexpr bool AreVectorsSequential(T first, const Args&... args) {
|
||||
return (fn(first, args) && ...);
|
||||
}
|
||||
|
||||
// Returns if the immediate can fit in to add/sub immediate instruction encodings.
|
||||
constexpr bool IsImmAddSub(uint64_t imm) {
|
||||
constexpr uint64_t U12Mask = 0xFFF;
|
||||
auto FitsWithin12Bits = [](uint64_t imm) {
|
||||
return (imm & ~U12Mask) == 0;
|
||||
};
|
||||
// Can fit in to the instruction encoding:
|
||||
// - if only bits [11:0] are set.
|
||||
// - if only bits [23:12] are set.
|
||||
return FitsWithin12Bits(imm) || (FitsWithin12Bits(imm >> 12) && (imm & U12Mask) == 0);
|
||||
}
|
||||
|
||||
// This is an emitter that is designed around the smallest code bloat as possible.
|
||||
// Eschewing most developer convenience in order to keep code as small as possible.
|
||||
|
||||
|
||||
@@ -2569,7 +2569,19 @@ public:
|
||||
}
|
||||
|
||||
// SVE contiguous non-temporal load (scalar plus immediate)
|
||||
// XXX:
|
||||
void ldnt1b(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalLoad(0b00, zt, pg, rn, Imm);
|
||||
}
|
||||
void ldnt1h(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalLoad(0b01, zt, pg, rn, Imm);
|
||||
}
|
||||
void ldnt1w(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalLoad(0b10, zt, pg, rn, Imm);
|
||||
}
|
||||
void ldnt1d(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalLoad(0b11, zt, pg, rn, Imm);
|
||||
}
|
||||
|
||||
// SVE contiguous non-temporal load (scalar plus scalar)
|
||||
// XXX:
|
||||
// SVE load multiple structures (scalar plus immediate)
|
||||
@@ -4492,6 +4504,22 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// SVE contiguous non-temporal load (scalar plus immediate)
|
||||
void SVEContiguousNontemporalLoad(uint32_t msz, ZRegister zt, PRegister pg, Register rn, int32_t imm) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_AA_FMT(imm >= -8 && imm <= 7,
|
||||
"Invalid loadstore offset ({}). Must be between [-8, 7]", imm);
|
||||
|
||||
const auto imm4 = static_cast<uint32_t>(imm) & 0xF;
|
||||
uint32_t Instr = 0b1010'0100'0000'0000'1110'0000'0000'0000;
|
||||
Instr |= msz << 23;
|
||||
Instr |= imm4 << 16;
|
||||
Instr |= pg.Idx() << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= zt.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// SVE contiguous non-temporal store (scalar plus immediate)
|
||||
void SVEContiguousNontemporalStore(uint32_t msz, ZRegister zt, PRegister pg, Register rn, int32_t imm) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
|
||||
@@ -36,9 +36,9 @@
|
||||
// to by n, imm_s and imm_r are undefined.
|
||||
static bool IsImmLogical(uint64_t value,
|
||||
unsigned width,
|
||||
unsigned* n,
|
||||
unsigned* imm_s,
|
||||
unsigned* imm_r) {
|
||||
unsigned* n = nullptr,
|
||||
unsigned* imm_s = nullptr,
|
||||
unsigned* imm_r = nullptr) {
|
||||
[[maybe_unused]] constexpr auto kBRegSize = 8;
|
||||
[[maybe_unused]] constexpr auto kHRegSize = 16;
|
||||
[[maybe_unused]] constexpr auto kSRegSize = 32;
|
||||
@@ -243,6 +243,46 @@ static bool IsImmLogical(uint64_t value,
|
||||
return true;
|
||||
}
|
||||
|
||||
static inline bool IsIntN(unsigned n, int64_t x) {
|
||||
if (n == 64) return true;
|
||||
int64_t limit = INT64_C(1) << (n - 1);
|
||||
return (-limit <= x) && (x < limit);
|
||||
}
|
||||
|
||||
static inline bool IsUintN(unsigned n, int64_t x) {
|
||||
// Convert to an unsigned integer to avoid implementation-defined behavior.
|
||||
return !(static_cast<uint64_t>(x) >> n);
|
||||
}
|
||||
|
||||
// clang-format off
|
||||
#define INT_1_TO_32_LIST(V) \
|
||||
V(1) V(2) V(3) V(4) V(5) V(6) V(7) V(8) \
|
||||
V(9) V(10) V(11) V(12) V(13) V(14) V(15) V(16) \
|
||||
V(17) V(18) V(19) V(20) V(21) V(22) V(23) V(24) \
|
||||
V(25) V(26) V(27) V(28) V(29) V(30) V(31) V(32)
|
||||
|
||||
#define INT_33_TO_63_LIST(V) \
|
||||
V(33) V(34) V(35) V(36) V(37) V(38) V(39) V(40) \
|
||||
V(41) V(42) V(43) V(44) V(45) V(46) V(47) V(48) \
|
||||
V(49) V(50) V(51) V(52) V(53) V(54) V(55) V(56) \
|
||||
V(57) V(58) V(59) V(60) V(61) V(62) V(63)
|
||||
|
||||
#define INT_1_TO_63_LIST(V) INT_1_TO_32_LIST(V) INT_33_TO_63_LIST(V)
|
||||
|
||||
// clang-format on
|
||||
|
||||
#define DECLARE_IS_INT_N(N) \
|
||||
static inline bool IsInt##N(int64_t x) { return IsIntN(N, x); }
|
||||
|
||||
#define DECLARE_IS_UINT_N(N) \
|
||||
static inline bool IsUint##N(int64_t x) { return IsUintN(N, x); }
|
||||
|
||||
INT_1_TO_63_LIST(DECLARE_IS_INT_N)
|
||||
INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
|
||||
#undef DECLARE_IS_INT_N
|
||||
#undef DECLARE_IS_UINT_N
|
||||
|
||||
private:
|
||||
|
||||
template <typename V>
|
||||
|
||||
@@ -13,5 +13,14 @@ function(GenBinFmt Name)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
endfunction()
|
||||
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
configure_file(FEX-x86.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf)
|
||||
configure_file(FEX-x86_64.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf)
|
||||
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
|
||||
else()
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
endif()
|
||||
@@ -0,0 +1 @@
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
@@ -6,4 +6,3 @@ mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xf
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
expose_interpreter optional
|
||||
@@ -0,0 +1 @@
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
@@ -6,4 +6,3 @@ mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xf
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
expose_interpreter optional
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 31aa7f634b...29f979ee5a.
Vendored
+1
-1
Submodule External/drm-headers updated: 34a20394f7...8efb6dc03f.
Vendored
+1
-1
Submodule External/fmt updated: f5e54359df...0c9fce2ffe.
Vendored
-1
Submodule External/imgui deleted from 4c986ecb8d.
Vendored
+1
-1
Submodule External/jemalloc updated: 5695452413...02ca52b5fe.
Vendored
+1
-1
Submodule External/jemalloc_glibc updated: 888181c5f7...404353974e.
Vendored
-1
Submodule External/json-maker deleted from 8ecb8ecc34.
Vendored
+1
-1
Submodule External/robin-map updated: f1ab690046...d5683d9f18.
Vendored
-1
Submodule External/tiny-json deleted from 9d09127f87.
Vendored
+3
@@ -0,0 +1,3 @@
|
||||
set(NAME tiny-json)
|
||||
set(SRCS tiny-json.c)
|
||||
add_library(${NAME} ${SRCS})
|
||||
Vendored
+21
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2018 Rafa Garcia
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Vendored
+647
@@ -0,0 +1,647 @@
|
||||
|
||||
/*
|
||||
|
||||
<https://github.com/rafagafe/tiny-json>
|
||||
|
||||
Licensed under the MIT License <http://opensource.org/licenses/MIT>.
|
||||
SPDX-License-Identifier: MIT
|
||||
Copyright (c) 2016-2018 Rafa Garcia <rafagarcia77@gmail.com>.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <ctype.h>
|
||||
#include <stddef.h> // For NULL
|
||||
#include "tiny-json.h"
|
||||
|
||||
/** Structure to handle a heap of JSON properties. */
|
||||
typedef struct jsonStaticPool_s {
|
||||
json_t* const mem; /**< Pointer to array of json properties. */
|
||||
unsigned int const qty; /**< Length of the array of json properties. */
|
||||
unsigned int nextFree; /**< The index of the next free json property. */
|
||||
jsonPool_t pool;
|
||||
} jsonStaticPool_t;
|
||||
|
||||
/* Search a property by its name in a JSON object. */
|
||||
json_t const* json_getProperty( json_t const* obj, char const* property ) {
|
||||
json_t const* sibling;
|
||||
for( sibling = obj->u.c.child; sibling; sibling = sibling->sibling )
|
||||
if ( sibling->name && !strcmp( sibling->name, property ) )
|
||||
return sibling;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Search a property by its name in a JSON object and return its value. */
|
||||
char const* json_getPropertyValue( json_t const* obj, char const* property ) {
|
||||
json_t const* field = json_getProperty( obj, property );
|
||||
if ( !field ) return 0;
|
||||
jsonType_t type = json_getType( field );
|
||||
if ( JSON_ARRAY >= type ) return 0;
|
||||
return json_getValue( field );
|
||||
}
|
||||
|
||||
/* Internal prototypes: */
|
||||
static char* goBlank( char* str );
|
||||
static char* goNum( char* str );
|
||||
static json_t* poolInit( jsonPool_t* pool );
|
||||
static json_t* poolAlloc( jsonPool_t* pool );
|
||||
static char* objValue( char* ptr, json_t* obj, jsonPool_t* pool );
|
||||
static char* setToNull( char* ch );
|
||||
static bool isEndOfPrimitive( char ch );
|
||||
|
||||
/* Parse a string to get a json. */
|
||||
json_t const* json_createWithPool( char *str, jsonPool_t *pool ) {
|
||||
char* ptr = goBlank( str );
|
||||
if ( !ptr || *ptr != '{' ) return 0;
|
||||
json_t* obj = pool->init( pool );
|
||||
obj->name = 0;
|
||||
obj->sibling = 0;
|
||||
obj->u.c.child = 0;
|
||||
ptr = objValue( ptr, obj, pool );
|
||||
if ( !ptr ) return 0;
|
||||
return obj;
|
||||
}
|
||||
|
||||
/* Parse a string to get a json. */
|
||||
json_t const* json_create( char* str, json_t mem[], unsigned int qty ) {
|
||||
jsonStaticPool_t spool = {
|
||||
.mem = mem,
|
||||
.qty = qty,
|
||||
.pool = {
|
||||
.init = poolInit,
|
||||
.alloc = poolAlloc
|
||||
}
|
||||
};
|
||||
return json_createWithPool( str, &spool.pool );
|
||||
}
|
||||
|
||||
/** Get a special character with its escape character. Examples:
|
||||
* 'b' -> '\b', 'n' -> '\n', 't' -> '\t'
|
||||
* @param ch The escape character.
|
||||
* @return The character code. */
|
||||
static char getEscape( char ch ) {
|
||||
static struct { char ch; char code; } const pair[] = {
|
||||
{ '\"', '\"' }, { '\\', '\\' },
|
||||
{ '/', '/' }, { 'b', '\b' },
|
||||
{ 'f', '\f' }, { 'n', '\n' },
|
||||
{ 'r', '\r' }, { 't', '\t' },
|
||||
};
|
||||
unsigned int i;
|
||||
for( i = 0; i < sizeof pair / sizeof *pair; ++i )
|
||||
if ( pair[i].ch == ch )
|
||||
return pair[i].code;
|
||||
return '\0';
|
||||
}
|
||||
|
||||
/** Parse 4 characters.
|
||||
* @Param str Pointer to first digit.
|
||||
* @retval '?' If the four characters are hexadecimal digits.
|
||||
* @retcal '\0' In other cases. */
|
||||
static unsigned char getCharFromUnicode( unsigned char const* str ) {
|
||||
unsigned int i;
|
||||
for( i = 0; i < 4; ++i )
|
||||
if ( !isxdigit( str[i] ) )
|
||||
return '\0';
|
||||
return '?';
|
||||
}
|
||||
|
||||
/** Parse a string and replace the scape characters by their meaning characters.
|
||||
* This parser stops when finds the character '\"'. Then replaces '\"' by '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* parseString( char* str ) {
|
||||
unsigned char* head = (unsigned char*)str;
|
||||
unsigned char* tail = (unsigned char*)str;
|
||||
for( ; *head >= ' '; ++head, ++tail ) {
|
||||
if ( *head == '\"' ) {
|
||||
*tail = '\0';
|
||||
return (char*)++head;
|
||||
}
|
||||
if ( *head == '\\' ) {
|
||||
if ( *++head == 'u' ) {
|
||||
char const ch = getCharFromUnicode( ++head );
|
||||
if ( ch == '\0' ) return 0;
|
||||
*tail = ch;
|
||||
head += 3;
|
||||
}
|
||||
else {
|
||||
char const esc = getEscape( *head );
|
||||
if ( esc == '\0' ) return 0;
|
||||
*tail = esc;
|
||||
}
|
||||
}
|
||||
else *tail = *head;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Parse a string to get the name of a property.
|
||||
* @param str Pointer to first character.
|
||||
* @param property The property to assign the name.
|
||||
* @retval Pointer to first of property value. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* propertyName( char* ptr, json_t* property ) {
|
||||
property->name = ++ptr;
|
||||
ptr = parseString( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
ptr = goBlank( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
if ( *ptr++ != ':' ) return 0;
|
||||
return goBlank( ptr );
|
||||
}
|
||||
|
||||
/** Parse a string to get the value of a property when its type is JSON_TEXT.
|
||||
* @param str Pointer to first character ('\"').
|
||||
* @param property The property to assign the name.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* textValue( char* ptr, json_t* property ) {
|
||||
++property->u.value;
|
||||
ptr = parseString( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_TEXT;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Compare two strings until get the null character in the second one.
|
||||
* @param ptr sub string
|
||||
* @param str main string
|
||||
* @retval Pointer to next character.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* checkStr( char* ptr, char const* str ) {
|
||||
while( *str )
|
||||
if ( *ptr++ != *str++ )
|
||||
return 0;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a primitive value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @param value String with the primitive literal.
|
||||
* @param type The code of the type. ( JSON_BOOLEAN or JSON_NULL )
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* primitiveValue( char* ptr, json_t* property, char const* value, jsonType_t type ) {
|
||||
ptr = checkStr( ptr, value );
|
||||
if ( !ptr || !isEndOfPrimitive( *ptr ) ) return 0;
|
||||
ptr = setToNull( ptr );
|
||||
property->type = type;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a true value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* trueValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "true", JSON_BOOLEAN );
|
||||
}
|
||||
|
||||
/** Parser a string to get a false value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* falseValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "false", JSON_BOOLEAN );
|
||||
}
|
||||
|
||||
/** Parser a string to get a null value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* nullValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "null", JSON_NULL );
|
||||
}
|
||||
|
||||
/** Analyze the exponential part of a real number.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non numerical after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* expValue( char* ptr ) {
|
||||
if ( *ptr == '-' || *ptr == '+' ) ++ptr;
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
ptr = goNum( ++ptr );
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Analyze the decimal part of a real number.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non numerical after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* fraqValue( char* ptr ) {
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
ptr = goNum( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a numerical value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type: JSON_REAL or JSON_INTEGER.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* numValue( char* ptr, json_t* property ) {
|
||||
if ( *ptr == '-' ) ++ptr;
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
if ( *ptr != '0' ) {
|
||||
ptr = goNum( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
else if ( isdigit( *++ptr ) ) return 0;
|
||||
property->type = JSON_INTEGER;
|
||||
if ( *ptr == '.' ) {
|
||||
ptr = fraqValue( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_REAL;
|
||||
}
|
||||
if ( *ptr == 'e' || *ptr == 'E' ) {
|
||||
ptr = expValue( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_REAL;
|
||||
}
|
||||
if ( !isEndOfPrimitive( *ptr ) ) return 0;
|
||||
if ( JSON_INTEGER == property->type ) {
|
||||
char const* value = property->u.value;
|
||||
bool const negative = *value == '-';
|
||||
static char const min[] = "-9223372036854775808";
|
||||
static char const max[] = "9223372036854775807";
|
||||
unsigned int const maxdigits = ( negative? sizeof min: sizeof max ) - 1;
|
||||
unsigned int const len = ptr - value;
|
||||
if ( len > maxdigits ) return 0;
|
||||
if ( len == maxdigits ) {
|
||||
char const tmp = *ptr;
|
||||
*ptr = '\0';
|
||||
char const* const threshold = negative ? min: max;
|
||||
if ( 0 > strcmp( threshold, value ) ) return 0;
|
||||
*ptr = tmp;
|
||||
}
|
||||
}
|
||||
ptr = setToNull( ptr );
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Add a property to a JSON object or array.
|
||||
* @param obj The handler of the JSON object or array.
|
||||
* @param property The handler of the property to be added. */
|
||||
static void add( json_t* obj, json_t* property ) {
|
||||
property->sibling = 0;
|
||||
if ( !obj->u.c.child ){
|
||||
obj->u.c.child = property;
|
||||
obj->u.c.last_child = property;
|
||||
} else {
|
||||
obj->u.c.last_child->sibling = property;
|
||||
obj->u.c.last_child = property;
|
||||
}
|
||||
}
|
||||
|
||||
/** Parser a string to get a json object value.
|
||||
* @param str Pointer to first character.
|
||||
* @param pool The handler of a json pool for creating json instances.
|
||||
* @retval Pointer to first character after the value. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* objValue( char* ptr, json_t* obj, jsonPool_t* pool ) {
|
||||
obj->type = JSON_OBJ;
|
||||
obj->u.c.child = 0;
|
||||
obj->sibling = 0;
|
||||
ptr++;
|
||||
for(;;) {
|
||||
ptr = goBlank( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
if ( *ptr == ',' ) {
|
||||
++ptr;
|
||||
continue;
|
||||
}
|
||||
char const endchar = ( obj->type == JSON_OBJ )? '}': ']';
|
||||
if ( *ptr == endchar ) {
|
||||
*ptr = '\0';
|
||||
json_t* parentObj = obj->sibling;
|
||||
if ( !parentObj ) return ++ptr;
|
||||
obj->sibling = 0;
|
||||
obj = parentObj;
|
||||
++ptr;
|
||||
continue;
|
||||
}
|
||||
json_t* property = pool->alloc( pool );
|
||||
if ( !property ) return 0;
|
||||
if( obj->type != JSON_ARRAY ) {
|
||||
if ( *ptr != '\"' ) return 0;
|
||||
ptr = propertyName( ptr, property );
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
else property->name = 0;
|
||||
add( obj, property );
|
||||
property->u.value = ptr;
|
||||
switch( *ptr ) {
|
||||
case '{':
|
||||
property->type = JSON_OBJ;
|
||||
property->u.c.child = 0;
|
||||
property->sibling = obj;
|
||||
obj = property;
|
||||
++ptr;
|
||||
break;
|
||||
case '[':
|
||||
property->type = JSON_ARRAY;
|
||||
property->u.c.child = 0;
|
||||
property->sibling = obj;
|
||||
obj = property;
|
||||
++ptr;
|
||||
break;
|
||||
case '\"': ptr = textValue( ptr, property ); break;
|
||||
case 't': ptr = trueValue( ptr, property ); break;
|
||||
case 'f': ptr = falseValue( ptr, property ); break;
|
||||
case 'n': ptr = nullValue( ptr, property ); break;
|
||||
default: ptr = numValue( ptr, property ); break;
|
||||
}
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Initialize a json pool.
|
||||
* @param pool The handler of the pool.
|
||||
* @return a instance of a json. */
|
||||
static json_t* poolInit( jsonPool_t* pool ) {
|
||||
jsonStaticPool_t *spool = json_containerOf( pool, jsonStaticPool_t, pool );
|
||||
spool->nextFree = 1;
|
||||
return spool->mem;
|
||||
}
|
||||
|
||||
/** Create an instance of a json from a pool.
|
||||
* @param pool The handler of the pool.
|
||||
* @retval The handler of the new instance if success.
|
||||
* @retval Null pointer if the pool was empty. */
|
||||
static json_t* poolAlloc( jsonPool_t* pool ) {
|
||||
jsonStaticPool_t *spool = json_containerOf( pool, jsonStaticPool_t, pool );
|
||||
if ( spool->nextFree >= spool->qty ) return 0;
|
||||
return spool->mem + spool->nextFree++;
|
||||
}
|
||||
|
||||
/** Checks whether an character belongs to set.
|
||||
* @param ch Character value to be checked.
|
||||
* @param set Set of characters. It is just a null-terminated string.
|
||||
* @return true or false there is membership or not. */
|
||||
static bool isOneOfThem( char ch, char const* set ) {
|
||||
while( *set != '\0' )
|
||||
if ( ch == *set++ )
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Increases a pointer while it points to a character that belongs to a set.
|
||||
* @param str The initial pointer value.
|
||||
* @param set Set of characters. It is just a null-terminated string.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goWhile( char* str, char const* set ) {
|
||||
for(; *str != '\0'; ++str ) {
|
||||
if ( !isOneOfThem( *str, set ) )
|
||||
return str;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Set of characters that defines a blank. */
|
||||
static char const* const blank = " \n\r\t\f";
|
||||
|
||||
/** Increases a pointer while it points to a white space character.
|
||||
* @param str The initial pointer value.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goBlank( char* str ) {
|
||||
return goWhile( str, blank );
|
||||
}
|
||||
|
||||
/** Increases a pointer while it points to a decimal digit character.
|
||||
* @param str The initial pointer value.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goNum( char* str ) {
|
||||
for( ; *str != '\0'; ++str ) {
|
||||
if ( !isdigit( *str ) )
|
||||
return str;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Set of characters that defines the end of an array or a JSON object. */
|
||||
static char const* const endofblock = "}]";
|
||||
|
||||
/** Set a char to '\0' and increase its pointer if the char is different to '}' or ']'.
|
||||
* @param ch Pointer to character.
|
||||
* @return Final value pointer. */
|
||||
static char* setToNull( char* ch ) {
|
||||
if ( !isOneOfThem( *ch, endofblock ) ) *ch++ = '\0';
|
||||
return ch;
|
||||
}
|
||||
|
||||
/** Indicate if a character is the end of a primitive value. */
|
||||
static bool isEndOfPrimitive( char ch ) {
|
||||
return ch == ',' || isOneOfThem( ch, blank ) || isOneOfThem( ch, endofblock );
|
||||
}
|
||||
|
||||
/** Add a character at the end of a string.
|
||||
* @param dest Pointer to the null character of the string
|
||||
* @param ch Value to be added.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* chtoa( char* dest, char ch ) {
|
||||
*dest = ch;
|
||||
*++dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Copy a null-terminated string.
|
||||
* @param dest Destination memory block.
|
||||
* @param src Source string.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* atoa( char* dest, char const* src ) {
|
||||
for( ; *src != '\0'; ++dest, ++src )
|
||||
*dest = *src;
|
||||
*dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Open a JSON object in a JSON string. */
|
||||
char* json_objOpen( char* dest, char const* name ) {
|
||||
if ( NULL == name )
|
||||
dest = chtoa( dest, '{' );
|
||||
else {
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":{" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Close a JSON object in a JSON string. */
|
||||
char* json_objClose( char* dest ) {
|
||||
if ( dest[-1] == ',' )
|
||||
--dest;
|
||||
return atoa( dest, "}," );
|
||||
}
|
||||
|
||||
/* Open an array in a JSON string. */
|
||||
char* json_arrOpen( char* dest, char const* name ) {
|
||||
if ( NULL == name )
|
||||
dest = chtoa( dest, '[' );
|
||||
else {
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":[" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Close an array in a JSON string. */
|
||||
char* json_arrClose( char* dest ) {
|
||||
if ( dest[-1] == ',' )
|
||||
--dest;
|
||||
return atoa( dest, "]," );
|
||||
}
|
||||
|
||||
/** Add the name of a text property.
|
||||
* @param dest Destination memory.
|
||||
* @param name The name of the property.
|
||||
* @return Pointer to the next char. */
|
||||
static char* strname( char* dest, char const* name ) {
|
||||
dest = chtoa( dest, '\"' );
|
||||
if ( NULL != name ) {
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":\"" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Get the hexadecimal digit of the least significant nibble of a integer. */
|
||||
static int nibbletoch( int nibble ) {
|
||||
return "0123456789ABCDEF"[ nibble % 16u ];
|
||||
}
|
||||
|
||||
/** Get the escape character of a non-printable.
|
||||
* @param ch Character source.
|
||||
* @return The escape character or null character if error. */
|
||||
static int escape( int ch ) {
|
||||
static struct { char code; char ch; } const pair[] = {
|
||||
{ '\"', '\"' }, { '\\', '\\' }, { '/', '/' }, { 'b', '\b' },
|
||||
{ 'f', '\f' }, { 'n', '\n' }, { 'r', '\r' }, { 't', '\t' },
|
||||
};
|
||||
for( int i = 0; i < sizeof pair / sizeof *pair; ++i )
|
||||
if ( ch == pair[i].ch )
|
||||
return pair[i].code;
|
||||
return '\0';
|
||||
}
|
||||
|
||||
/** Copy a null-terminated string inserting escape characters if needed.
|
||||
* @param dest Destination memory block.
|
||||
* @param src Source string.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* atoesc( char* dest, char const* src ) {
|
||||
for( ; *src != '\0'; ++dest, ++src ) {
|
||||
if ( *src >= ' ' && *src != '\"' && *src != '\\' && *src != '/' )
|
||||
*dest = *src;
|
||||
else {
|
||||
*dest++ = '\\';
|
||||
int const esc = escape( *src );
|
||||
if ( esc )
|
||||
*dest = esc;
|
||||
else {
|
||||
*dest++ = 'u';
|
||||
*dest++ = '0';
|
||||
*dest++ = '0';
|
||||
*dest++ = nibbletoch( *src / 16 );
|
||||
*dest++ = nibbletoch( *src );
|
||||
}
|
||||
}
|
||||
}
|
||||
*dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a text property in a JSON string. */
|
||||
char* json_str( char* dest, char const* name, char const* value ) {
|
||||
dest = strname( dest, name );
|
||||
dest = atoesc( dest, value );
|
||||
dest = atoa( dest, "\"," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Add the name of a primitive property.
|
||||
* @param dest Destination memory.
|
||||
* @param name The name of the property.
|
||||
* @return Pointer to the next char. */
|
||||
static char* primitivename( char* dest, char const* name ) {
|
||||
if( NULL == name )
|
||||
return dest;
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":" );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a boolean property in a JSON string. */
|
||||
char* json_bool( char* dest, char const* name, int value ) {
|
||||
dest = primitivename( dest, name );
|
||||
dest = atoa( dest, value ? "true," : "false," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a null property in a JSON string. */
|
||||
char* json_null( char* dest, char const* name ) {
|
||||
dest = primitivename( dest, name );
|
||||
dest = atoa( dest, "null," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Used to finish the root JSON object. After call json_objClose(). */
|
||||
char* json_end( char* dest ) {
|
||||
if ( ',' == dest[-1] ) {
|
||||
dest[-1] = '\0';
|
||||
--dest;
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
#define ALL_TYPES \
|
||||
X( json_int, int, "%d" ) \
|
||||
X( json_long, long, "%ld" ) \
|
||||
X( json_uint, unsigned int, "%u" ) \
|
||||
X( json_ulong, unsigned long, "%lu" ) \
|
||||
X( json_verylong, long long, "%lld" ) \
|
||||
X( json_double, double, "%g" ) \
|
||||
|
||||
|
||||
#define json_num( funcname, type, fmt ) \
|
||||
char* funcname( char* dest, char const* name, type value ) { \
|
||||
dest = primitivename( dest, name ); \
|
||||
dest += sprintf( dest, fmt, value ); \
|
||||
dest = chtoa( dest, ',' ); \
|
||||
return dest; \
|
||||
}
|
||||
|
||||
#define X( name, type, fmt ) json_num( name, type, fmt )
|
||||
ALL_TYPES
|
||||
#undef X
|
||||
Vendored
+270
@@ -0,0 +1,270 @@
|
||||
|
||||
/*
|
||||
|
||||
<https://github.com/rafagafe/tiny-json>
|
||||
|
||||
Licensed under the MIT License <http://opensource.org/licenses/MIT>.
|
||||
SPDX-License-Identifier: MIT
|
||||
Copyright (c) 2016-2018 Rafa Garcia <rafagarcia77@gmail.com>.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
*/
|
||||
|
||||
#ifndef _TINY_JSON_H_
|
||||
#define _TINY_JSON_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#define json_containerOf( ptr, type, member ) \
|
||||
((type*)( (char*)ptr - offsetof( type, member ) ))
|
||||
|
||||
/** @defgroup tinyJson Tiny JSON parser.
|
||||
* @{ */
|
||||
|
||||
/** Enumeration of codes of supported JSON properties types. */
|
||||
typedef enum {
|
||||
JSON_OBJ, JSON_ARRAY, JSON_TEXT, JSON_BOOLEAN,
|
||||
JSON_INTEGER, JSON_REAL, JSON_NULL
|
||||
} jsonType_t;
|
||||
|
||||
/** Structure to handle JSON properties. */
|
||||
typedef struct json_s {
|
||||
struct json_s* sibling;
|
||||
const char* name;
|
||||
union {
|
||||
const char* value;
|
||||
struct {
|
||||
struct json_s* child;
|
||||
struct json_s* last_child;
|
||||
} c;
|
||||
} u;
|
||||
jsonType_t type;
|
||||
} json_t;
|
||||
|
||||
/** Parse a string to get a json.
|
||||
* @param str String pointer with a JSON object. It will be modified.
|
||||
* @param mem Array of json properties to allocate.
|
||||
* @param qty Number of elements of mem.
|
||||
* @retval Null pointer if any was wrong in the parse process.
|
||||
* @retval If the parser process was successfully a valid handler of a json.
|
||||
* This property is always unnamed and its type is JSON_OBJ. */
|
||||
const json_t* json_create(char* str, json_t mem[], unsigned int qty);
|
||||
|
||||
/** Get the name of a json property.
|
||||
* @param json A valid handler of a json property.
|
||||
* @retval Pointer to null-terminated if property has name.
|
||||
* @retval Null pointer if the property is unnamed. */
|
||||
static inline const char* json_getName(const json_t* json) {
|
||||
return json->name;
|
||||
}
|
||||
|
||||
/** Get the value of a json property.
|
||||
* The type of property cannot be JSON_OBJ or JSON_ARRAY.
|
||||
* @param json A valid handler of a json property.
|
||||
* @return Pointer to null-terminated string with the value. */
|
||||
static inline const char* json_getValue(const json_t* property) {
|
||||
return property->u.value;
|
||||
}
|
||||
|
||||
/** Get the type of a json property.
|
||||
* @param json A valid handler of a json property.
|
||||
* @return The code of type.*/
|
||||
static inline jsonType_t json_getType(const json_t* json) {
|
||||
return json->type;
|
||||
}
|
||||
|
||||
/** Get the next sibling of a JSON property that is within a JSON object or array.
|
||||
* @param json A valid handler of a json property.
|
||||
* @retval The handler of the next sibling if found.
|
||||
* @retval Null pointer if the json property is the last one. */
|
||||
static inline const json_t* json_getSibling(const json_t* json) {
|
||||
return json->sibling;
|
||||
}
|
||||
|
||||
/** Search a property by its name in a JSON object.
|
||||
* @param obj A valid handler of a json object. Its type must be JSON_OBJ.
|
||||
* @param property The name of property to get.
|
||||
* @retval The handler of the json property if found.
|
||||
* @retval Null pointer if not found. */
|
||||
const json_t* json_getProperty(const json_t* obj, const char* property);
|
||||
|
||||
|
||||
/** Search a property by its name in a JSON object and return its value.
|
||||
* @param obj A valid handler of a json object. Its type must be JSON_OBJ.
|
||||
* @param property The name of property to get.
|
||||
* @retval If found a pointer to null-terminated string with the value.
|
||||
* @retval Null pointer if not found or it is an array or an object. */
|
||||
const char* json_getPropertyValue(const json_t* obj, const char* property);
|
||||
|
||||
/** Get the first property of a JSON object or array.
|
||||
* @param json A valid handler of a json property.
|
||||
* Its type must be JSON_OBJ or JSON_ARRAY.
|
||||
* @retval The handler of the first property if there is.
|
||||
* @retval Null pointer if the json object has not properties. */
|
||||
static inline const json_t* json_getChild(const json_t* json) {
|
||||
return json->u.c.child;
|
||||
}
|
||||
|
||||
/** Get the value of a json boolean property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_BOOLEAN.
|
||||
* @return The value stdbool. */
|
||||
static inline bool json_getBoolean(const json_t* property) {
|
||||
return *property->u.value == 't';
|
||||
}
|
||||
|
||||
/** Get the value of a json integer property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_INTEGER.
|
||||
* @return The value stdint. */
|
||||
static inline int64_t json_getInteger(const json_t* property) {
|
||||
return atoll( property->u.value );
|
||||
}
|
||||
|
||||
/** Get the value of a json real property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_REAL.
|
||||
* @return The value. */
|
||||
static inline double json_getReal(const json_t* property) {
|
||||
return atof( property->u.value );
|
||||
}
|
||||
|
||||
|
||||
/** Structure to handle a heap of JSON properties. */
|
||||
typedef struct jsonPool_s jsonPool_t;
|
||||
struct jsonPool_s {
|
||||
json_t* (*init)( jsonPool_t* pool );
|
||||
json_t* (*alloc)( jsonPool_t* pool );
|
||||
};
|
||||
|
||||
/** Parse a string to get a json.
|
||||
* @param str String pointer with a JSON object. It will be modified.
|
||||
* @param pool Custom json pool pointer.
|
||||
* @retval Null pointer if any was wrong in the parse process.
|
||||
* @retval If the parser process was successfully a valid handler of a json.
|
||||
* This property is always unnamed and its type is JSON_OBJ. */
|
||||
const json_t* json_createWithPool(char* str, jsonPool_t* pool);
|
||||
|
||||
/** @ } */
|
||||
|
||||
/** @defgroup makejoson Make JSON.
|
||||
* @{ */
|
||||
|
||||
/** Open a JSON object in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_objOpen(char* dest, const char* name);
|
||||
|
||||
/** Close a JSON object in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_objClose(char* dest);
|
||||
|
||||
/** Used to finish the root JSON object. After call json_objClose().
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_end(char* dest);
|
||||
|
||||
/** Open an array in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_arrOpen(char* dest, const char* name);
|
||||
|
||||
/** Close an array in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_arrClose(char* dest);
|
||||
|
||||
/** Add a text property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value A valid null-terminated string with the value.
|
||||
* Backslash escapes will be added for special characters.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_str(char* dest, const char* name, const char* value);
|
||||
|
||||
/** Add a boolean property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Zero for false. Non zero for true.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_bool(char* dest, const char* name, int value);
|
||||
|
||||
/** Add a null property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_null(char* dest, const char* name);
|
||||
|
||||
/** Add an integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_int(char* dest, const char* name, int value);
|
||||
|
||||
/** Add an unsigned integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_uint(char* dest, const char* name, unsigned int value);
|
||||
|
||||
/** Add a long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_long(char* dest, const char* name, long int value);
|
||||
|
||||
/** Add an unsigned long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_ulong(char* dest, const char* name, unsigned long int value);
|
||||
|
||||
/** Add a long long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_verylong(char* dest, const char* name, long long int value);
|
||||
|
||||
/** Add a double precision number property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_double(char* dest, const char* name, double value);
|
||||
|
||||
/** @ } */
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* _TINY_JSON_H_ */
|
||||
@@ -24,27 +24,6 @@ include(CheckCXXCompilerFlag)
|
||||
include(CheckIncludeFileCXX)
|
||||
include(CheckCXXSourceCompiles)
|
||||
|
||||
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
__attribute__((preserve_all))
|
||||
int Testy(int a, int b, int c, int d, int e, int f) {
|
||||
return a + b + c + d + e + f;
|
||||
}
|
||||
int main() {
|
||||
return Testy(0, 1, 2, 3, 4, 5);
|
||||
}"
|
||||
HAS_CLANG_PRESERVE_ALL)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
if (HAS_CLANG_PRESERVE_ALL)
|
||||
if (MINGW_BUILD)
|
||||
message(STATUS "Ignoring broken clang::preserve_all support")
|
||||
set(HAS_CLANG_PRESERVE_ALL FALSE)
|
||||
else()
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
# Useful to have for freestanding libFEXCore
|
||||
add_subdirectory(External/vixl/)
|
||||
|
||||
@@ -148,9 +148,8 @@ def print_man_options(options):
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
|
||||
output_man.write("\n.sp\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -179,9 +178,8 @@ def print_man_environment(options):
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
|
||||
output_man.write("\n.sp\n")
|
||||
|
||||
print_man_environment_tail()
|
||||
output_man.write(".El\n")
|
||||
@@ -219,6 +217,14 @@ def print_man_environment_tail():
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored. These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
@@ -395,7 +401,7 @@ def print_parse_argloader_options(options):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "("
|
||||
conversion_func = "std::move("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
import json
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
import textwrap
|
||||
|
||||
def ExitError(msg):
|
||||
print(msg)
|
||||
@@ -53,6 +54,7 @@ class OpDefinition:
|
||||
SSAArgNum: int
|
||||
NonSSAArgNum: int
|
||||
DynamicDispatch: bool
|
||||
LoweredX87: bool
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
@@ -76,6 +78,7 @@ class OpDefinition:
|
||||
self.SSAArgNum = 0
|
||||
self.NonSSAArgNum = 0
|
||||
self.DynamicDispatch = False
|
||||
self.LoweredX87 = False
|
||||
self.JITDispatch = True
|
||||
self.JITDispatchOverride = None
|
||||
self.TiedSource = -1
|
||||
@@ -122,21 +125,36 @@ def parse_ops(ops):
|
||||
|
||||
RHS = EqualSplit[0].strip()
|
||||
if len(EqualSplit) > 1:
|
||||
OpDef.HasDest = True
|
||||
LHS = EqualSplit[0].strip()
|
||||
RHS = EqualSplit[1].strip()
|
||||
|
||||
# Parse the destination, must be one type of SSA, GPR, or FPR
|
||||
ResultType = EqualSplit[0].strip()
|
||||
if ResultType == "SSA":
|
||||
OpDef.DestType = "SSA" # We don't know this type right now
|
||||
elif ResultType == "GPR":
|
||||
OpDef.DestType = "GPR"
|
||||
elif ResultType == "GPRPair":
|
||||
OpDef.DestType = "GPRPair"
|
||||
elif ResultType == "FPR":
|
||||
OpDef.DestType = "FPR"
|
||||
if ":" in LHS:
|
||||
# Named destinations. This is a hack, but so is the entire
|
||||
# multi-destination support bolten onto the old IR...
|
||||
#
|
||||
# Named destinations require side effects because they break
|
||||
# SSA hard. Validate that.
|
||||
assert("HasSideEffects" in op_val and op_val["HasSideEffects"])
|
||||
|
||||
for Dest in LHS.split(","):
|
||||
Dest = Dest.strip()
|
||||
DType, Name = Dest.split(":$")
|
||||
|
||||
# If the destination appears also as a source, it is
|
||||
# read-modify-write.
|
||||
if Dest in RHS:
|
||||
# Turn RMW into an in/out source
|
||||
RHS = RHS.replace(Dest.strip(), f"{DType}:$Inout{Name}")
|
||||
else:
|
||||
# Turn named destinations into an out source.
|
||||
RHS += f", {DType}:$Out{Name}"
|
||||
else:
|
||||
ExitError("Unknown destination class type {}. Needs to be one of {SSA, GPR, GPRPair, FPR}".format(ResultType))
|
||||
# Single anonymous destination
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR")
|
||||
|
||||
OpDef.HasDest = True
|
||||
OpDef.DestType = LHS
|
||||
|
||||
# IR Op needs to start with a name
|
||||
RHS = RHS.split(" ", 1)
|
||||
@@ -204,7 +222,7 @@ def parse_ops(ops):
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpDef.EmitValidation.append("GetOpRegClass({}) == InvalidClass || WalkFindRegClass({}) == {}Class".format(NameWithPrefix, NameWithPrefix, OpArg.Type))
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
OpArg.NameWithPrefix = NameWithPrefix
|
||||
@@ -250,6 +268,13 @@ def parse_ops(ops):
|
||||
if "JITDispatchOverride" in op_val:
|
||||
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
|
||||
|
||||
if "X87" in op_val:
|
||||
OpDef.LoweredX87 = op_val["X87"]
|
||||
|
||||
# X87 implies !JITDispatch
|
||||
assert("JITDispatch" not in op_val)
|
||||
OpDef.JITDispatch = False
|
||||
|
||||
if "TiedSource" in op_val:
|
||||
OpDef.TiedSource = op_val["TiedSource"]
|
||||
|
||||
@@ -258,12 +283,8 @@ def parse_ops(ops):
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
# Patch up all the argument names
|
||||
for Arg in OpDef.Arguments:
|
||||
if Arg.Temporary:
|
||||
# Temporary ops just replace all instances no prefix variant
|
||||
OpDef.EmitValidation[i] = OpDef.EmitValidation[i].replace(Arg.NameWithPrefix, Arg.Name)
|
||||
else:
|
||||
# All other ops replace $ with _ variant for argument passed in
|
||||
OpDef.EmitValidation[i] = OpDef.EmitValidation[i].replace(Arg.NameWithPrefix, "_{}".format(Arg.Name))
|
||||
# Temporary ops just replace all instances no prefix variant
|
||||
OpDef.EmitValidation[i] = OpDef.EmitValidation[i].replace(Arg.NameWithPrefix, Arg.Name)
|
||||
|
||||
#OpDef.print()
|
||||
|
||||
@@ -368,42 +389,28 @@ def print_ir_sizes():
|
||||
if op.Name == "Last":
|
||||
output_file.write("\t-1ULL,\n")
|
||||
else:
|
||||
output_file.write("\tsizeof(IROp_{}),\n".format(op.Name))
|
||||
output_file.write(f"\tsizeof(IROp_{op.Name}),\n")
|
||||
|
||||
output_file.write("};\n\n")
|
||||
output_file.write(textwrap.dedent("""
|
||||
};
|
||||
|
||||
output_file.write("// Make sure our array maps directly to the IROps enum\n")
|
||||
output_file.write("static_assert(IRSizes[IROps::OP_LAST] == -1ULL);\n\n")
|
||||
// Make sure our array maps directly to the IROps enum
|
||||
static_assert(IRSizes[IROps::OP_LAST] == -1ULL);
|
||||
|
||||
output_file.write("[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
|
||||
[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool LoweredX87(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);
|
||||
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);\n'
|
||||
)
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
output_file.write("#endif\n\n")
|
||||
#undef IROP_SIZES
|
||||
#endif
|
||||
"""))
|
||||
|
||||
def print_ir_reg_classes():
|
||||
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
|
||||
@@ -493,13 +500,14 @@ def print_ir_getraargs():
|
||||
def print_ir_hassideeffects():
|
||||
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
|
||||
for array, prop, T in [
|
||||
("SideEffects", "HasSideEffects", "bool"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber", "bool"),
|
||||
("TiedSources", "TiedSource", "int8_t"),
|
||||
for prop, T in [
|
||||
("HasSideEffects", "bool"),
|
||||
("ImplicitFlagClobber", "bool"),
|
||||
("LoweredX87", "bool"),
|
||||
("TiedSource", "int8_t"),
|
||||
]:
|
||||
output_file.write(
|
||||
f"constexpr std::array<{'uint8_t' if T == 'bool' else T}, OP_LAST + 1> {array} = {{\n"
|
||||
f"constexpr std::array<{'uint8_t' if T == 'bool' else T}, OP_LAST + 1> {prop}_ = {{\n"
|
||||
)
|
||||
for op in IROps:
|
||||
if T == "bool":
|
||||
@@ -512,7 +520,7 @@ def print_ir_hassideeffects():
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write(f"{T} {prop}(IROps Op) {{\n")
|
||||
output_file.write(f" return {array}[Op];\n")
|
||||
output_file.write(f" return {prop}_[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
@@ -671,11 +679,11 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNode *_{}".format(arg.Name))
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} _{}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
@@ -689,23 +697,28 @@ def print_ir_allocator_helpers():
|
||||
if op.ImplicitFlagClobber:
|
||||
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
|
||||
|
||||
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
# We gather the "has x87?" flag as we go. This saves the user from
|
||||
# having to keep track of whether they emitted any x87.
|
||||
if op.LoweredX87:
|
||||
output_file.write("\t\tRecordX87Use();\n")
|
||||
|
||||
output_file.write("\t\tauto _Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tOp.first->{} = _{}->Wrapped(ListDataBegin);\n".format(arg.Name, arg.Name))
|
||||
output_file.write("\t\t_Op.first->{} = {}->Wrapped(ListDataBegin);\n".format(arg.Name, arg.Name))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t_{}->AddUse();\n".format(arg.Name))
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if not arg.Temporary and not arg.IsSSA:
|
||||
output_file.write("\t\tOp.first->{} = _{};\n".format(arg.Name, arg.Name))
|
||||
output_file.write("\t\t_Op.first->{} = {};\n".format(arg.Name, arg.Name))
|
||||
|
||||
if (op.HasDest):
|
||||
# We can only infer a size if we have arguments
|
||||
@@ -715,22 +728,22 @@ def print_ir_allocator_helpers():
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tuint8_t Size{} = GetOpSize(_{});\n".format(arg.Name, arg.Name))
|
||||
output_file.write("\t\tuint8_t Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tInferSize = std::max(InferSize, Size{});\n".format(arg.Name))
|
||||
|
||||
output_file.write("\t\tOp.first->Header.Size = InferSize;\n")
|
||||
output_file.write("\t\t_Op.first->Header.Size = InferSize;\n")
|
||||
|
||||
# Some ops without a destination still need an operating size
|
||||
# Effectively reusing the destination size value for operation size
|
||||
if op.DestSize != None:
|
||||
output_file.write("\t\tOp.first->Header.Size = {};\n".format(op.DestSize))
|
||||
output_file.write("\t\t_Op.first->Header.Size = {};\n".format(op.DestSize))
|
||||
|
||||
if op.NumElements == None:
|
||||
output_file.write("\t\tOp.first->Header.ElementSize = Op.first->Header.Size / ({});\n".format(1))
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(1))
|
||||
else:
|
||||
output_file.write("\t\tOp.first->Header.ElementSize = Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
@@ -741,58 +754,12 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
output_file.write("\t\treturn _Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
def print_ir_parser_switch_helper():
|
||||
output_file.write("#ifdef IROP_PARSER_SWITCH_HELPERS\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.SwitchGen:
|
||||
output_file.write("\tcase FEXCore::IR::IROps::OP_%s: {\n" % (op.Name.upper()))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("\t\tauto arg{} = DecodeValue<{}>(Def.Args[{}]);\n".format(i, CType, i))
|
||||
output_file.write("\t\tif (!CheckPrintErrorArg(Def, arg{}.first, {})) return false;\n".format(i, i))
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("\t\tauto arg{} = DecodeValue<OrderedNode*>(Def.Args[{}]);\n".format(i, i))
|
||||
output_file.write("\t\tif (!CheckPrintErrorArg(Def, arg{}.first, {})) return false;\n".format(i, i))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("\t\tauto arg{} = DecodeValue<{}>(Def.Args[{}]);\n".format(i, CType, i))
|
||||
output_file.write("\t\tif (!CheckPrintErrorArg(Def, arg{}.first, {})) return false;\n".format(i, i))
|
||||
|
||||
output_file.write("\t\tDef.Node = _{}(\n".format(op.Name))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
output_file.write("\t\t\targ{}.second".format(i))
|
||||
if not LastArg:
|
||||
output_file.write(",\n")
|
||||
else:
|
||||
output_file.write("\n")
|
||||
|
||||
output_file.write("\t\t);\n")
|
||||
|
||||
output_file.write("\t\tSSANameMapper[Def.Definition] = Def.Node;\n")
|
||||
|
||||
output_file.write("\t\tbreak;\n")
|
||||
output_file.write("\t}\n")
|
||||
|
||||
|
||||
output_file.write("#undef IROP_PARSER_SWITCH_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
def print_ir_dispatcher_defs():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DEFS\n")
|
||||
for op in IROps:
|
||||
@@ -851,7 +818,6 @@ print_ir_hassideeffects()
|
||||
print_ir_gethasdest()
|
||||
print_ir_arg_printer()
|
||||
print_ir_allocator_helpers()
|
||||
print_ir_parser_switch_helper()
|
||||
|
||||
output_file.close()
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
include(GNUInstallDirs)
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
@@ -67,7 +68,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt_1Ks.c
|
||||
Common/SoftFloat-3e/softfloat_raiseFlags.c
|
||||
Common/SoftFloat-3e/softfloat_state.c
|
||||
Common/SoftFloat-3e/f64_to_extF80.c
|
||||
Common/SoftFloat-3e/s_commonNaNToExtF80UI.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF64Sig.c
|
||||
@@ -86,12 +86,10 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
@@ -113,7 +111,6 @@ set (SRCS
|
||||
Interface/Core/JIT/Arm64/BranchOps.cpp
|
||||
Interface/Core/JIT/Arm64/ConversionOps.cpp
|
||||
Interface/Core/JIT/Arm64/EncryptionOps.cpp
|
||||
Interface/Core/JIT/Arm64/FlagOps.cpp
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
@@ -121,7 +118,6 @@ set (SRCS
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
Interface/Core/X86Tables/H0F38Tables.cpp
|
||||
Interface/Core/X86Tables/H0F3ATables.cpp
|
||||
Interface/Core/X86Tables/PrimaryGroupTables.cpp
|
||||
@@ -131,20 +127,18 @@ set (SRCS
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/DeadContextStoreElimination.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
@@ -161,7 +155,7 @@ if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
Utils/AllocatorOverride.cpp)
|
||||
endif()
|
||||
|
||||
set(DEFINES -DTHREAD_LOCAL=_Thread_local -DJIT_ARM64)
|
||||
set(DEFINES -DJIT_ARM64)
|
||||
|
||||
if (_M_X86_64)
|
||||
list(APPEND DEFINES -D_M_X86_64=1)
|
||||
@@ -171,11 +165,6 @@ if (_M_ARM_64)
|
||||
list(APPEND DEFINES -D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# We can run the simulator on both x86-64 or AArch64 hosts
|
||||
list(APPEND DEFINES -DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER)
|
||||
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
|
||||
endif()
|
||||
@@ -189,7 +178,11 @@ endif()
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
|
||||
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
list (APPEND LIBS vixl)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
@@ -200,14 +193,6 @@ else()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
@@ -377,11 +362,30 @@ AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
install(TARGETS ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
|
||||
# Meta-library to link jemalloc libraries enabled in the build configuration.
|
||||
# Only needed for targets that run emulation. For others, use JemallocDummy.
|
||||
add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
|
||||
if (ENABLE_JEMALLOC)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_JEMALLOC=1 JEMALLOC_NO_RENAME=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC FEX_jemalloc)
|
||||
endif()
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
|
||||
target_link_libraries(JemallocLibs INTERFACE FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
# Dummy project to use for host tools.
|
||||
# This overrides use of jemalloc in FEXCore with the normal glibc allocator.
|
||||
add_library(JemallocDummy STATIC Utils/AllocatorHooks.cpp)
|
||||
target_include_directories(JemallocDummy PRIVATE "${PROJECT_SOURCE_DIR}/include/")
|
||||
endif()
|
||||
|
||||
# The shared library should always link enabled jemalloc libraries
|
||||
target_link_libraries(${PROJECT_NAME}_shared JemallocLibs)
|
||||
@@ -55,7 +55,7 @@ void JITSymbols::RegisterJITSpace(const void* HostAddr, uint32_t CodeSize) {
|
||||
}
|
||||
|
||||
// Buffered JIT symbols.
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
void JITSymbols::Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -79,7 +79,7 @@ void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, u
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
void JITSymbols::Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -104,7 +104,7 @@ void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, u
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
void JITSymbols::RegisterNamedRegion(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
@@ -128,7 +128,7 @@ void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void*
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
void JITSymbols::WriteBuffer(FEXCore::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
auto Now = std::chrono::steady_clock::now();
|
||||
if (!ForceWrite) {
|
||||
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) && Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
|
||||
|
||||
@@ -11,6 +11,26 @@
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
constexpr static std::chrono::milliseconds MAXIMUM_THRESHOLD {100};
|
||||
|
||||
JITSymbolBuffer()
|
||||
: LastWrite {std::chrono::steady_clock::now()} {}
|
||||
// stead_clock to ensure a monotonic increasing clock.
|
||||
// In highly stressed situations this can still cause >2% CPU time in vdso_clock_gettime.
|
||||
// If we need lower CPU time when JIT symbols are enabled then FEX can read the cycle counter directly.
|
||||
std::chrono::steady_clock::time_point LastWrite {};
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
JITSymbols();
|
||||
@@ -21,16 +41,16 @@ public:
|
||||
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
|
||||
|
||||
// Allocate JIT buffer.
|
||||
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<Core::JITSymbolBuffer>();
|
||||
static fextl::unique_ptr<FEXCore::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<FEXCore::JITSymbolBuffer>();
|
||||
}
|
||||
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(FEXCore::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
|
||||
private:
|
||||
int fd {-1};
|
||||
void WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
void WriteBuffer(FEXCore::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -41,7 +41,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_add( extFloat80_t a, extFloat80_t b )
|
||||
extFloat80_t extF80_add( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -53,7 +53,7 @@ extFloat80_t extF80_add( extFloat80_t a, extFloat80_t b )
|
||||
bool signB;
|
||||
extFloat80_t
|
||||
(*magsFuncPtr)(
|
||||
uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
struct softfloat_state *, uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
|
||||
uA.f = a;
|
||||
uiA64 = uA.s.signExp;
|
||||
@@ -65,6 +65,6 @@ extFloat80_t extF80_add( extFloat80_t a, extFloat80_t b )
|
||||
signB = signExtF80UI64( uiB64 );
|
||||
magsFuncPtr =
|
||||
(signA == signB) ? softfloat_addMagsExtF80 : softfloat_subMagsExtF80;
|
||||
return (*magsFuncPtr)( uiA64, uiA0, uiB64, uiB0, signA );
|
||||
return (*magsFuncPtr)( state, uiA64, uiA0, uiB64, uiB0, signA );
|
||||
}
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_div( extFloat80_t a, extFloat80_t b )
|
||||
extFloat80_t extF80_div( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -107,7 +107,7 @@ extFloat80_t extF80_div( extFloat80_t a, extFloat80_t b )
|
||||
if ( ! (sigB & UINT64_C( 0x8000000000000000 )) ) {
|
||||
if ( ! sigB ) {
|
||||
if ( ! sigA ) goto invalid;
|
||||
softfloat_raiseFlags( softfloat_flag_infinite );
|
||||
softfloat_raiseFlags( state, softfloat_flag_infinite );
|
||||
goto infinity;
|
||||
}
|
||||
normExpSig = softfloat_normSubnormalExtF80Sig( sigB );
|
||||
@@ -169,18 +169,18 @@ extFloat80_t extF80_div( extFloat80_t a, extFloat80_t b )
|
||||
sigZExtra = (uint64_t) ((uint_fast64_t) q<<41);
|
||||
return
|
||||
softfloat_roundPackToExtF80(
|
||||
signZ, expZ, sigZ, sigZExtra, extF80_roundingPrecision );
|
||||
state, signZ, expZ, sigZ, sigZExtra, state->roundingPrecision );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
propagateNaN:
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
goto uiZ;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
invalid:
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
uiZ64 = defaultNaNExtF80UI64;
|
||||
uiZ0 = defaultNaNExtF80UI0;
|
||||
goto uiZ;
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
bool extF80_eq( extFloat80_t a, extFloat80_t b )
|
||||
bool extF80_eq( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -62,7 +62,7 @@ bool extF80_eq( extFloat80_t a, extFloat80_t b )
|
||||
softfloat_isSigNaNExtF80UI( uiA64, uiA0 )
|
||||
|| softfloat_isSigNaNExtF80UI( uiB64, uiB0 )
|
||||
) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
bool extF80_lt( extFloat80_t a, extFloat80_t b )
|
||||
bool extF80_lt( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -59,7 +59,7 @@ bool extF80_lt( extFloat80_t a, extFloat80_t b )
|
||||
uiB64 = uB.s.signExp;
|
||||
uiB0 = uB.s.signif;
|
||||
if ( isNaNExtF80UI( uiA64, uiA0 ) || isNaNExtF80UI( uiB64, uiB0 ) ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return false;
|
||||
}
|
||||
signA = signExtF80UI64( uiA64 );
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_mul( extFloat80_t a, extFloat80_t b )
|
||||
extFloat80_t extF80_mul( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -125,11 +125,11 @@ extFloat80_t extF80_mul( extFloat80_t a, extFloat80_t b )
|
||||
}
|
||||
return
|
||||
softfloat_roundPackToExtF80(
|
||||
signZ, expZ, sig128Z.v64, sig128Z.v0, extF80_roundingPrecision );
|
||||
state, signZ, expZ, sig128Z.v64, sig128Z.v0, state->roundingPrecision );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
propagateNaN:
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
goto uiZ;
|
||||
@@ -137,7 +137,7 @@ extFloat80_t extF80_mul( extFloat80_t a, extFloat80_t b )
|
||||
*------------------------------------------------------------------------*/
|
||||
infArg:
|
||||
if ( ! magBits ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
uiZ64 = defaultNaNExtF80UI64;
|
||||
uiZ0 = defaultNaNExtF80UI0;
|
||||
} else {
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_rem( extFloat80_t a, extFloat80_t b )
|
||||
extFloat80_t extF80_rem( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -193,18 +193,18 @@ extFloat80_t extF80_rem( extFloat80_t a, extFloat80_t b )
|
||||
}
|
||||
return
|
||||
softfloat_normRoundPackToExtF80(
|
||||
signRem, rem.v64 | rem.v0 ? expB + 32 : 0, rem.v64, rem.v0, 80 );
|
||||
state, signRem, rem.v64 | rem.v0 ? expB + 32 : 0, rem.v64, rem.v0, 80 );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
propagateNaN:
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
goto uiZ;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
invalid:
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
uiZ64 = defaultNaNExtF80UI64;
|
||||
uiZ0 = defaultNaNExtF80UI0;
|
||||
goto uiZ;
|
||||
|
||||
@@ -43,7 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t
|
||||
extF80_roundToInt( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
extF80_roundToInt( struct softfloat_state *state, extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64, signUI64;
|
||||
@@ -80,7 +80,7 @@ extFloat80_t
|
||||
if ( 0x403E <= exp ) {
|
||||
if ( exp == 0x7FFF ) {
|
||||
if ( sigA & UINT64_C( 0x7FFFFFFFFFFFFFFF ) ) {
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, sigA, 0, 0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, sigA, 0, 0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
sigZ = uiZ.v0;
|
||||
goto uiZ;
|
||||
@@ -93,7 +93,7 @@ extFloat80_t
|
||||
goto uiZ;
|
||||
}
|
||||
if ( exp <= 0x3FFE ) {
|
||||
if ( exact ) softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( exact ) state->exceptionFlags |= softfloat_flag_inexact;
|
||||
switch ( roundingMode ) {
|
||||
case softfloat_round_near_even:
|
||||
if ( !(sigA & UINT64_C( 0x7FFFFFFFFFFFFFFF )) ) break;
|
||||
@@ -145,7 +145,7 @@ extFloat80_t
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) sigZ |= lastBitMask;
|
||||
#endif
|
||||
if ( exact ) softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( exact ) state->exceptionFlags |= softfloat_flag_inexact;
|
||||
}
|
||||
uiZ:
|
||||
uZ.s.signExp = uiZ64;
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_sqrt( extFloat80_t a )
|
||||
extFloat80_t extF80_sqrt( struct softfloat_state *state, extFloat80_t a )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -74,7 +74,7 @@ extFloat80_t extF80_sqrt( extFloat80_t a )
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( expA == 0x7FFF ) {
|
||||
if ( sigA & UINT64_C( 0x7FFFFFFFFFFFFFFF ) ) {
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, uiA0, 0, 0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, uiA0, 0, 0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
goto uiZ;
|
||||
@@ -155,11 +155,11 @@ extFloat80_t extF80_sqrt( extFloat80_t a )
|
||||
}
|
||||
return
|
||||
softfloat_roundPackToExtF80(
|
||||
0, expZ, sigZ, sigZExtra, extF80_roundingPrecision );
|
||||
state, 0, expZ, sigZ, sigZExtra, state->roundingPrecision );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
invalid:
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
uiZ64 = defaultNaNExtF80UI64;
|
||||
uiZ0 = defaultNaNExtF80UI0;
|
||||
goto uiZ;
|
||||
|
||||
@@ -41,7 +41,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_sub( extFloat80_t a, extFloat80_t b )
|
||||
extFloat80_t extF80_sub( struct softfloat_state *state, extFloat80_t a, extFloat80_t b )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -54,7 +54,7 @@ extFloat80_t extF80_sub( extFloat80_t a, extFloat80_t b )
|
||||
#if ! defined INLINE_LEVEL || (INLINE_LEVEL < 2)
|
||||
extFloat80_t
|
||||
(*magsFuncPtr)(
|
||||
uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
struct softfloat_state *, uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
#endif
|
||||
|
||||
uA.f = a;
|
||||
@@ -67,14 +67,14 @@ extFloat80_t extF80_sub( extFloat80_t a, extFloat80_t b )
|
||||
signB = signExtF80UI64( uiB64 );
|
||||
#if defined INLINE_LEVEL && (2 <= INLINE_LEVEL)
|
||||
if ( signA == signB ) {
|
||||
return softfloat_subMagsExtF80( uiA64, uiA0, uiB64, uiB0, signA );
|
||||
return softfloat_subMagsExtF80( state, uiA64, uiA0, uiB64, uiB0, signA );
|
||||
} else {
|
||||
return softfloat_addMagsExtF80( uiA64, uiA0, uiB64, uiB0, signA );
|
||||
return softfloat_addMagsExtF80( state, uiA64, uiA0, uiB64, uiB0, signA );
|
||||
}
|
||||
#else
|
||||
magsFuncPtr =
|
||||
(signA == signB) ? softfloat_subMagsExtF80 : softfloat_addMagsExtF80;
|
||||
return (*magsFuncPtr)( uiA64, uiA0, uiB64, uiB0, signA );
|
||||
return (*magsFuncPtr)( state, uiA64, uiA0, uiB64, uiB0, signA );
|
||||
#endif
|
||||
|
||||
}
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float128_t extF80_to_f128( extFloat80_t a )
|
||||
float128_t extF80_to_f128( struct softfloat_state *state, extFloat80_t a )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -61,7 +61,7 @@ float128_t extF80_to_f128( extFloat80_t a )
|
||||
exp = expExtF80UI64( uiA64 );
|
||||
frac = uiA0 & UINT64_C( 0x7FFFFFFFFFFFFFFF );
|
||||
if ( (exp == 0x7FFF) && frac ) {
|
||||
softfloat_extF80UIToCommonNaN( uiA64, uiA0, &commonNaN );
|
||||
softfloat_extF80UIToCommonNaN( state, uiA64, uiA0, &commonNaN );
|
||||
uiZ = softfloat_commonNaNToF128UI( &commonNaN );
|
||||
} else {
|
||||
sign = signExtF80UI64( uiA64 );
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float32_t extF80_to_f32( extFloat80_t a )
|
||||
float32_t extF80_to_f32( struct softfloat_state *state, extFloat80_t a )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -66,7 +66,7 @@ float32_t extF80_to_f32( extFloat80_t a )
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( exp == 0x7FFF ) {
|
||||
if ( sig & UINT64_C( 0x7FFFFFFFFFFFFFFF ) ) {
|
||||
softfloat_extF80UIToCommonNaN( uiA64, uiA0, &commonNaN );
|
||||
softfloat_extF80UIToCommonNaN( state, uiA64, uiA0, &commonNaN );
|
||||
uiZ = softfloat_commonNaNToF32UI( &commonNaN );
|
||||
} else {
|
||||
uiZ = packToF32UI( sign, 0xFF, 0 );
|
||||
@@ -86,7 +86,7 @@ float32_t extF80_to_f32( extFloat80_t a )
|
||||
if ( sizeof (int_fast16_t) < sizeof (int_fast32_t) ) {
|
||||
if ( exp < -0x1000 ) exp = -0x1000;
|
||||
}
|
||||
return softfloat_roundPackToF32( sign, exp, sig32 );
|
||||
return softfloat_roundPackToF32( state, sign, exp, sig32 );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
uiZ:
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float64_t extF80_to_f64( extFloat80_t a )
|
||||
float64_t extF80_to_f64( struct softfloat_state *state, extFloat80_t a )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -72,7 +72,7 @@ float64_t extF80_to_f64( extFloat80_t a )
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( exp == 0x7FFF ) {
|
||||
if ( sig & UINT64_C( 0x7FFFFFFFFFFFFFFF ) ) {
|
||||
softfloat_extF80UIToCommonNaN( uiA64, uiA0, &commonNaN );
|
||||
softfloat_extF80UIToCommonNaN( state, uiA64, uiA0, &commonNaN );
|
||||
uiZ = softfloat_commonNaNToF64UI( &commonNaN );
|
||||
} else {
|
||||
uiZ = packToF64UI( sign, 0x7FF, 0 );
|
||||
@@ -86,7 +86,7 @@ float64_t extF80_to_f64( extFloat80_t a )
|
||||
if ( sizeof (int_fast16_t) < sizeof (int_fast32_t) ) {
|
||||
if ( exp < -0x1000 ) exp = -0x1000;
|
||||
}
|
||||
return softfloat_roundPackToF64( sign, exp, sig );
|
||||
return softfloat_roundPackToF64( state, sign, exp, sig );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
uiZ:
|
||||
|
||||
@@ -43,7 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast32_t
|
||||
extF80_to_i32( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
extF80_to_i32( struct softfloat_state *state, extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -68,7 +68,7 @@ int_fast32_t
|
||||
#elif (i32_fromNaN == i32_fromNegOverflow)
|
||||
sign = 1;
|
||||
#else
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return i32_fromNaN;
|
||||
#endif
|
||||
}
|
||||
@@ -78,7 +78,7 @@ int_fast32_t
|
||||
shiftDist = 0x4032 - exp;
|
||||
if ( shiftDist <= 0 ) shiftDist = 1;
|
||||
sig = softfloat_shiftRightJam64( sig, shiftDist );
|
||||
return softfloat_roundToI32( sign, sig, roundingMode, exact );
|
||||
return softfloat_roundToI32( state, sign, sig, roundingMode, exact );
|
||||
|
||||
}
|
||||
|
||||
@@ -43,7 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast64_t
|
||||
extF80_to_i64( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
extF80_to_i64( struct softfloat_state *state, extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -68,7 +68,7 @@ int_fast64_t
|
||||
/*--------------------------------------------------------------------
|
||||
*--------------------------------------------------------------------*/
|
||||
if ( shiftDist ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return
|
||||
(exp == 0x7FFF) && (sig & UINT64_C( 0x7FFFFFFFFFFFFFFF ))
|
||||
? i64_fromNaN
|
||||
@@ -84,7 +84,7 @@ int_fast64_t
|
||||
sig = sig64Extra.v;
|
||||
sigExtra = sig64Extra.extra;
|
||||
}
|
||||
return softfloat_roundToI64( sign, sig, sigExtra, roundingMode, exact );
|
||||
return softfloat_roundToI64( state, sign, sig, sigExtra, roundingMode, exact );
|
||||
|
||||
}
|
||||
|
||||
@@ -43,7 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
uint_fast64_t
|
||||
extF80_to_ui64( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
extF80_to_ui64( struct softfloat_state *state, extFloat80_t a, uint_fast8_t roundingMode, bool exact )
|
||||
{
|
||||
union { struct extFloat80M s; extFloat80_t f; } uA;
|
||||
uint_fast16_t uiA64;
|
||||
@@ -65,7 +65,7 @@ uint_fast64_t
|
||||
*------------------------------------------------------------------------*/
|
||||
shiftDist = 0x403E - exp;
|
||||
if ( shiftDist < 0 ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return
|
||||
(exp == 0x7FFF) && (sig & UINT64_C( 0x7FFFFFFFFFFFFFFF ))
|
||||
? ui64_fromNaN
|
||||
@@ -79,7 +79,7 @@ uint_fast64_t
|
||||
sig = sig64Extra.v;
|
||||
sigExtra = sig64Extra.extra;
|
||||
}
|
||||
return softfloat_roundToUI64( sign, sig, sigExtra, roundingMode, exact );
|
||||
return softfloat_roundToUI64( state, sign, sig, sigExtra, roundingMode, exact );
|
||||
|
||||
}
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t f128_to_extF80( float128_t a )
|
||||
extFloat80_t f128_to_extF80( struct softfloat_state *state, float128_t a )
|
||||
{
|
||||
union ui128_f128 uA;
|
||||
uint_fast64_t uiA64, uiA0;
|
||||
@@ -70,7 +70,7 @@ extFloat80_t f128_to_extF80( float128_t a )
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( exp == 0x7FFF ) {
|
||||
if ( frac64 | frac0 ) {
|
||||
softfloat_f128UIToCommonNaN( uiA64, uiA0, &commonNaN );
|
||||
softfloat_f128UIToCommonNaN( state, uiA64, uiA0, &commonNaN );
|
||||
uiZ = softfloat_commonNaNToExtF80UI( &commonNaN );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
@@ -98,7 +98,7 @@ extFloat80_t f128_to_extF80( float128_t a )
|
||||
sig128 =
|
||||
softfloat_shortShiftLeft128(
|
||||
frac64 | UINT64_C( 0x0001000000000000 ), frac0, 15 );
|
||||
return softfloat_roundPackToExtF80( sign, exp, sig128.v64, sig128.v0, 80 );
|
||||
return softfloat_roundPackToExtF80( state, sign, exp, sig128.v64, sig128.v0, 80 );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
uiZ:
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t f32_to_extF80( float32_t a )
|
||||
extFloat80_t f32_to_extF80( struct softfloat_state *state, float32_t a )
|
||||
{
|
||||
union ui32_f32 uA;
|
||||
uint_fast32_t uiA;
|
||||
@@ -67,7 +67,7 @@ extFloat80_t f32_to_extF80( float32_t a )
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( exp == 0xFF ) {
|
||||
if ( frac ) {
|
||||
softfloat_f32UIToCommonNaN( uiA, &commonNaN );
|
||||
softfloat_f32UIToCommonNaN( state, uiA, &commonNaN );
|
||||
uiZ = softfloat_commonNaNToExtF80UI( &commonNaN );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include "softfloat.h"
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t f64_to_extF80( float64_t a )
|
||||
extFloat80_t f64_to_extF80( struct softfloat_state *state, float64_t a )
|
||||
{
|
||||
union ui64_f64 uA;
|
||||
uint_fast64_t uiA;
|
||||
@@ -67,7 +67,7 @@ extFloat80_t f64_to_extF80( float64_t a )
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( exp == 0x7FF ) {
|
||||
if ( frac ) {
|
||||
softfloat_f64UIToCommonNaN( uiA, &commonNaN );
|
||||
softfloat_f64UIToCommonNaN( state, uiA, &commonNaN );
|
||||
uiZ = softfloat_commonNaNToExtF80UI( &commonNaN );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
|
||||
@@ -63,19 +63,19 @@ uint_fast32_t softfloat_roundToUI32( bool, uint_fast64_t, uint_fast8_t, bool );
|
||||
#ifdef SOFTFLOAT_FAST_INT64
|
||||
uint_fast64_t
|
||||
softfloat_roundToUI64(
|
||||
bool, uint_fast64_t, uint_fast64_t, uint_fast8_t, bool );
|
||||
struct softfloat_state *, bool, uint_fast64_t, uint_fast64_t, uint_fast8_t, bool );
|
||||
#else
|
||||
uint_fast64_t softfloat_roundMToUI64( bool, uint32_t *, uint_fast8_t, bool );
|
||||
#endif
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast32_t softfloat_roundToI32( bool, uint_fast64_t, uint_fast8_t, bool );
|
||||
int_fast32_t softfloat_roundToI32( struct softfloat_state *, bool, uint_fast64_t, uint_fast8_t, bool );
|
||||
|
||||
#ifdef SOFTFLOAT_FAST_INT64
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast64_t
|
||||
softfloat_roundToI64(
|
||||
bool, uint_fast64_t, uint_fast64_t, uint_fast8_t, bool );
|
||||
struct softfloat_state *, bool, uint_fast64_t, uint_fast64_t, uint_fast8_t, bool );
|
||||
#else
|
||||
int_fast64_t softfloat_roundMToI64( bool, uint32_t *, uint_fast8_t, bool );
|
||||
#endif
|
||||
@@ -115,7 +115,7 @@ FEXCORE_PRESERVE_ALL_ATTR
|
||||
struct exp16_sig32 softfloat_normSubnormalF32Sig( uint_fast32_t );
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float32_t softfloat_roundPackToF32( bool, int_fast16_t, uint_fast32_t );
|
||||
float32_t softfloat_roundPackToF32( struct softfloat_state *, bool, int_fast16_t, uint_fast32_t );
|
||||
float32_t softfloat_normRoundPackToF32( bool, int_fast16_t, uint_fast32_t );
|
||||
|
||||
float32_t softfloat_addMagsF32( uint_fast32_t, uint_fast32_t );
|
||||
@@ -138,7 +138,7 @@ FEXCORE_PRESERVE_ALL_ATTR
|
||||
struct exp16_sig64 softfloat_normSubnormalF64Sig( uint_fast64_t );
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float64_t softfloat_roundPackToF64( bool, int_fast16_t, uint_fast64_t );
|
||||
float64_t softfloat_roundPackToF64( struct softfloat_state *, bool, int_fast16_t, uint_fast64_t );
|
||||
float64_t softfloat_normRoundPackToF64( bool, int_fast16_t, uint_fast64_t );
|
||||
|
||||
float64_t softfloat_addMagsF64( uint_fast64_t, uint_fast64_t, bool );
|
||||
@@ -167,18 +167,18 @@ struct exp32_sig64 softfloat_normSubnormalExtF80Sig( uint_fast64_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t
|
||||
softfloat_roundPackToExtF80(
|
||||
bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
|
||||
struct softfloat_state *, bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t
|
||||
softfloat_normRoundPackToExtF80(
|
||||
bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
|
||||
struct softfloat_state *, bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
|
||||
|
||||
extFloat80_t
|
||||
softfloat_addMagsExtF80(
|
||||
uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
struct softfloat_state *, uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
extFloat80_t
|
||||
softfloat_subMagsExtF80(
|
||||
uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
struct softfloat_state *, uint_fast16_t, uint_fast64_t, uint_fast16_t, uint_fast64_t, bool );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
*----------------------------------------------------------------------------*/
|
||||
|
||||
@@ -43,6 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
extFloat80_t
|
||||
softfloat_addMagsExtF80(
|
||||
struct softfloat_state *state,
|
||||
uint_fast16_t uiA64,
|
||||
uint_fast64_t uiA0,
|
||||
uint_fast16_t uiB64,
|
||||
@@ -140,11 +141,11 @@ extFloat80_t
|
||||
roundAndPack:
|
||||
return
|
||||
softfloat_roundPackToExtF80(
|
||||
signZ, expZ, sigZ, sigZExtra, extF80_roundingPrecision );
|
||||
state, signZ, expZ, sigZ, sigZExtra, state->roundingPrecision );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
propagateNaN:
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
uiZ:
|
||||
|
||||
@@ -49,11 +49,11 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void
|
||||
softfloat_extF80UIToCommonNaN(
|
||||
uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
|
||||
struct softfloat_state *state, uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
|
||||
{
|
||||
|
||||
if ( softfloat_isSigNaNExtF80UI( uiA64, uiA0 ) ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
}
|
||||
zPtr->sign = uiA64>>15;
|
||||
zPtr->v64 = uiA0<<1;
|
||||
|
||||
@@ -50,12 +50,12 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void
|
||||
softfloat_f128UIToCommonNaN(
|
||||
uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
|
||||
struct softfloat_state *state, uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
|
||||
{
|
||||
struct uint128 NaNSig;
|
||||
|
||||
if ( softfloat_isSigNaNF128UI( uiA64, uiA0 ) ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
}
|
||||
NaNSig = softfloat_shortShiftLeft128( uiA64, uiA0, 16 );
|
||||
zPtr->sign = uiA64>>63;
|
||||
|
||||
@@ -46,11 +46,11 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
| exception is raised.
|
||||
*----------------------------------------------------------------------------*/
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void softfloat_f32UIToCommonNaN( uint_fast32_t uiA, struct commonNaN *zPtr )
|
||||
void softfloat_f32UIToCommonNaN( struct softfloat_state *state, uint_fast32_t uiA, struct commonNaN *zPtr )
|
||||
{
|
||||
|
||||
if ( softfloat_isSigNaNF32UI( uiA ) ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
}
|
||||
zPtr->sign = uiA>>31;
|
||||
zPtr->v64 = (uint_fast64_t) uiA<<41;
|
||||
|
||||
@@ -46,11 +46,11 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
| exception is raised.
|
||||
*----------------------------------------------------------------------------*/
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void softfloat_f64UIToCommonNaN( uint_fast64_t uiA, struct commonNaN *zPtr )
|
||||
void softfloat_f64UIToCommonNaN( struct softfloat_state *state, uint_fast64_t uiA, struct commonNaN *zPtr )
|
||||
{
|
||||
|
||||
if ( softfloat_isSigNaNF64UI( uiA ) ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
}
|
||||
zPtr->sign = uiA>>63;
|
||||
zPtr->v64 = uiA<<12;
|
||||
|
||||
@@ -42,6 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t
|
||||
softfloat_normRoundPackToExtF80(
|
||||
struct softfloat_state *state,
|
||||
bool sign,
|
||||
int_fast32_t exp,
|
||||
uint_fast64_t sig,
|
||||
@@ -66,7 +67,7 @@ extFloat80_t
|
||||
}
|
||||
return
|
||||
softfloat_roundPackToExtF80(
|
||||
sign, exp, sig, sigExtra, roundingPrecision );
|
||||
state, sign, exp, sig, sigExtra, roundingPrecision );
|
||||
|
||||
}
|
||||
|
||||
@@ -53,6 +53,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
struct uint128
|
||||
softfloat_propagateNaNExtF80UI(
|
||||
struct softfloat_state *state,
|
||||
uint_fast16_t uiA64,
|
||||
uint_fast64_t uiA0,
|
||||
uint_fast16_t uiB64,
|
||||
@@ -76,7 +77,7 @@ struct uint128
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( isSigNaNA | isSigNaNB ) {
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
if ( isSigNaNA ) {
|
||||
if ( isSigNaNB ) goto returnLargerMag;
|
||||
if ( isNaNExtF80UI( uiB64, uiB0 ) ) goto returnB;
|
||||
|
||||
@@ -43,6 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t
|
||||
softfloat_roundPackToExtF80(
|
||||
struct softfloat_state *state,
|
||||
bool sign,
|
||||
int_fast32_t exp,
|
||||
uint_fast64_t sig,
|
||||
@@ -59,7 +60,7 @@ extFloat80_t
|
||||
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
roundingMode = softfloat_roundingMode;
|
||||
roundingMode = state->roundingMode;
|
||||
roundNearEven = (roundingMode == softfloat_round_near_even);
|
||||
if ( roundingPrecision == 80 ) goto precision80;
|
||||
if ( roundingPrecision == 64 ) {
|
||||
@@ -87,15 +88,15 @@ extFloat80_t
|
||||
/*----------------------------------------------------------------
|
||||
*----------------------------------------------------------------*/
|
||||
isTiny =
|
||||
(softfloat_detectTininess
|
||||
(state->detectTininess
|
||||
== softfloat_tininess_beforeRounding)
|
||||
|| (exp < 0)
|
||||
|| (sig <= (uint64_t) (sig + roundIncrement));
|
||||
sig = softfloat_shiftRightJam64( sig, 1 - exp );
|
||||
roundBits = sig & roundMask;
|
||||
if ( roundBits ) {
|
||||
if ( isTiny ) softfloat_raiseFlags( softfloat_flag_underflow );
|
||||
softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( isTiny ) softfloat_raiseFlags( state, softfloat_flag_underflow );
|
||||
state->exceptionFlags |= softfloat_flag_inexact;
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) {
|
||||
sig |= roundMask + 1;
|
||||
@@ -121,7 +122,7 @@ extFloat80_t
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( roundBits ) {
|
||||
softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
state->exceptionFlags |= softfloat_flag_inexact;
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) {
|
||||
sig = (sig & ~roundMask) | (roundMask + 1);
|
||||
@@ -157,7 +158,7 @@ extFloat80_t
|
||||
/*----------------------------------------------------------------
|
||||
*----------------------------------------------------------------*/
|
||||
isTiny =
|
||||
(softfloat_detectTininess
|
||||
(state->detectTininess
|
||||
== softfloat_tininess_beforeRounding)
|
||||
|| (exp < 0)
|
||||
|| ! doIncrement
|
||||
@@ -168,8 +169,8 @@ extFloat80_t
|
||||
sig = sig64Extra.v;
|
||||
sigExtra = sig64Extra.extra;
|
||||
if ( sigExtra ) {
|
||||
if ( isTiny ) softfloat_raiseFlags( softfloat_flag_underflow );
|
||||
softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( isTiny ) softfloat_raiseFlags( state, softfloat_flag_underflow );
|
||||
state->exceptionFlags |= softfloat_flag_inexact;
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) {
|
||||
sig |= 1;
|
||||
@@ -207,7 +208,7 @@ extFloat80_t
|
||||
roundMask = 0;
|
||||
overflow:
|
||||
softfloat_raiseFlags(
|
||||
softfloat_flag_overflow | softfloat_flag_inexact );
|
||||
state, softfloat_flag_overflow | softfloat_flag_inexact );
|
||||
if (
|
||||
roundNearEven
|
||||
|| (roundingMode == softfloat_round_near_maxMag)
|
||||
@@ -226,7 +227,7 @@ extFloat80_t
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
if ( sigExtra ) {
|
||||
softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
state->exceptionFlags |= softfloat_flag_inexact;
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) {
|
||||
sig |= 1;
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float32_t
|
||||
softfloat_roundPackToF32( bool sign, int_fast16_t exp, uint_fast32_t sig )
|
||||
softfloat_roundPackToF32( struct softfloat_state *state, bool sign, int_fast16_t exp, uint_fast32_t sig )
|
||||
{
|
||||
uint_fast8_t roundingMode;
|
||||
bool roundNearEven;
|
||||
@@ -53,7 +53,7 @@ float32_t
|
||||
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
roundingMode = softfloat_roundingMode;
|
||||
roundingMode = state->roundingMode;
|
||||
roundNearEven = (roundingMode == softfloat_round_near_even);
|
||||
roundIncrement = 0x40;
|
||||
if ( ! roundNearEven && (roundingMode != softfloat_round_near_maxMag) ) {
|
||||
@@ -71,19 +71,19 @@ float32_t
|
||||
/*----------------------------------------------------------------
|
||||
*----------------------------------------------------------------*/
|
||||
isTiny =
|
||||
(softfloat_detectTininess == softfloat_tininess_beforeRounding)
|
||||
(state->detectTininess == softfloat_tininess_beforeRounding)
|
||||
|| (exp < -1) || (sig + roundIncrement < 0x80000000);
|
||||
sig = softfloat_shiftRightJam32( sig, -exp );
|
||||
exp = 0;
|
||||
roundBits = sig & 0x7F;
|
||||
if ( isTiny && roundBits ) {
|
||||
softfloat_raiseFlags( softfloat_flag_underflow );
|
||||
softfloat_raiseFlags( state, softfloat_flag_underflow );
|
||||
}
|
||||
} else if ( (0xFD < exp) || (0x80000000 <= sig + roundIncrement) ) {
|
||||
/*----------------------------------------------------------------
|
||||
*----------------------------------------------------------------*/
|
||||
softfloat_raiseFlags(
|
||||
softfloat_flag_overflow | softfloat_flag_inexact );
|
||||
state, softfloat_flag_overflow | softfloat_flag_inexact );
|
||||
uiZ = packToF32UI( sign, 0xFF, 0 ) - ! roundIncrement;
|
||||
goto uiZ;
|
||||
}
|
||||
@@ -92,7 +92,7 @@ float32_t
|
||||
*------------------------------------------------------------------------*/
|
||||
sig = (sig + roundIncrement)>>7;
|
||||
if ( roundBits ) {
|
||||
softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
state->exceptionFlags |= softfloat_flag_inexact;
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) {
|
||||
sig |= 1;
|
||||
|
||||
@@ -42,7 +42,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float64_t
|
||||
softfloat_roundPackToF64( bool sign, int_fast16_t exp, uint_fast64_t sig )
|
||||
softfloat_roundPackToF64( struct softfloat_state *state, bool sign, int_fast16_t exp, uint_fast64_t sig )
|
||||
{
|
||||
uint_fast8_t roundingMode;
|
||||
bool roundNearEven;
|
||||
@@ -53,7 +53,7 @@ float64_t
|
||||
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
roundingMode = softfloat_roundingMode;
|
||||
roundingMode = state->roundingMode;
|
||||
roundNearEven = (roundingMode == softfloat_round_near_even);
|
||||
roundIncrement = 0x200;
|
||||
if ( ! roundNearEven && (roundingMode != softfloat_round_near_maxMag) ) {
|
||||
@@ -71,14 +71,14 @@ float64_t
|
||||
/*----------------------------------------------------------------
|
||||
*----------------------------------------------------------------*/
|
||||
isTiny =
|
||||
(softfloat_detectTininess == softfloat_tininess_beforeRounding)
|
||||
(state->detectTininess == softfloat_tininess_beforeRounding)
|
||||
|| (exp < -1)
|
||||
|| (sig + roundIncrement < UINT64_C( 0x8000000000000000 ));
|
||||
sig = softfloat_shiftRightJam64( sig, -exp );
|
||||
exp = 0;
|
||||
roundBits = sig & 0x3FF;
|
||||
if ( isTiny && roundBits ) {
|
||||
softfloat_raiseFlags( softfloat_flag_underflow );
|
||||
softfloat_raiseFlags( state, softfloat_flag_underflow );
|
||||
}
|
||||
} else if (
|
||||
(0x7FD < exp)
|
||||
@@ -87,7 +87,7 @@ float64_t
|
||||
/*----------------------------------------------------------------
|
||||
*----------------------------------------------------------------*/
|
||||
softfloat_raiseFlags(
|
||||
softfloat_flag_overflow | softfloat_flag_inexact );
|
||||
state, softfloat_flag_overflow | softfloat_flag_inexact );
|
||||
uiZ = packToF64UI( sign, 0x7FF, 0 ) - ! roundIncrement;
|
||||
goto uiZ;
|
||||
}
|
||||
@@ -96,7 +96,7 @@ float64_t
|
||||
*------------------------------------------------------------------------*/
|
||||
sig = (sig + roundIncrement)>>10;
|
||||
if ( roundBits ) {
|
||||
softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
state->exceptionFlags |= softfloat_flag_inexact;
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) {
|
||||
sig |= 1;
|
||||
|
||||
@@ -44,7 +44,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast32_t
|
||||
softfloat_roundToI32(
|
||||
bool sign, uint_fast64_t sig, uint_fast8_t roundingMode, bool exact )
|
||||
struct softfloat_state *state, bool sign, uint_fast64_t sig, uint_fast8_t roundingMode, bool exact )
|
||||
{
|
||||
uint_fast16_t roundIncrement, roundBits;
|
||||
uint_fast32_t sig32;
|
||||
@@ -86,13 +86,13 @@ int_fast32_t
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) z |= 1;
|
||||
#endif
|
||||
if ( exact ) softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( exact ) state->exceptionFlags |= softfloat_flag_inexact;
|
||||
}
|
||||
return z;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
invalid:
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return sign ? i32_fromNegOverflow : i32_fromPosOverflow;
|
||||
|
||||
}
|
||||
|
||||
@@ -44,6 +44,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast64_t
|
||||
softfloat_roundToI64(
|
||||
struct softfloat_state *state,
|
||||
bool sign,
|
||||
uint_fast64_t sig,
|
||||
uint_fast64_t sigExtra,
|
||||
@@ -89,13 +90,13 @@ int_fast64_t
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) z |= 1;
|
||||
#endif
|
||||
if ( exact ) softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( exact ) state->exceptionFlags |= softfloat_flag_inexact;
|
||||
}
|
||||
return z;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
invalid:
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return sign ? i64_fromNegOverflow : i64_fromPosOverflow;
|
||||
|
||||
}
|
||||
|
||||
@@ -43,6 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
uint_fast64_t
|
||||
softfloat_roundToUI64(
|
||||
struct softfloat_state *state,
|
||||
bool sign,
|
||||
uint_fast64_t sig,
|
||||
uint_fast64_t sigExtra,
|
||||
@@ -84,13 +85,13 @@ uint_fast64_t
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
if ( roundingMode == softfloat_round_odd ) sig |= 1;
|
||||
#endif
|
||||
if ( exact ) softfloat_exceptionFlags |= softfloat_flag_inexact;
|
||||
if ( exact ) state->exceptionFlags |= softfloat_flag_inexact;
|
||||
}
|
||||
return sig;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
invalid:
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
return sign ? ui64_fromNegOverflow : ui64_fromPosOverflow;
|
||||
|
||||
}
|
||||
|
||||
@@ -43,6 +43,7 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
extFloat80_t
|
||||
softfloat_subMagsExtF80(
|
||||
struct softfloat_state *state,
|
||||
uint_fast16_t uiA64,
|
||||
uint_fast64_t uiA0,
|
||||
uint_fast16_t uiB64,
|
||||
@@ -77,7 +78,7 @@ extFloat80_t
|
||||
if ( (sigA | sigB) & UINT64_C( 0x7FFFFFFFFFFFFFFF ) ) {
|
||||
goto propagateNaN;
|
||||
}
|
||||
softfloat_raiseFlags( softfloat_flag_invalid );
|
||||
softfloat_raiseFlags( state, softfloat_flag_invalid );
|
||||
uiZ64 = defaultNaNExtF80UI64;
|
||||
uiZ0 = defaultNaNExtF80UI0;
|
||||
goto uiZ;
|
||||
@@ -90,7 +91,7 @@ extFloat80_t
|
||||
if ( sigB < sigA ) goto aBigger;
|
||||
if ( sigA < sigB ) goto bBigger;
|
||||
uiZ64 =
|
||||
packToExtF80UI64( (softfloat_roundingMode == softfloat_round_min), 0 );
|
||||
packToExtF80UI64( (state->roundingMode == softfloat_round_min), 0 );
|
||||
uiZ0 = 0;
|
||||
goto uiZ;
|
||||
/*------------------------------------------------------------------------
|
||||
@@ -142,11 +143,11 @@ extFloat80_t
|
||||
normRoundPack:
|
||||
return
|
||||
softfloat_normRoundPackToExtF80(
|
||||
signZ, expZ, sig128.v64, sig128.v0, extF80_roundingPrecision );
|
||||
state, signZ, expZ, sig128.v64, sig128.v0, state->roundingPrecision );
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
propagateNaN:
|
||||
uiZ = softfloat_propagateNaNExtF80UI( uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ = softfloat_propagateNaNExtF80UI( state, uiA64, uiA0, uiB64, uiB0 );
|
||||
uiZ64 = uiZ.v64;
|
||||
uiZ0 = uiZ.v0;
|
||||
uiZ:
|
||||
|
||||
@@ -50,50 +50,11 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#include <stdint.h>
|
||||
#include "softfloat_types.h"
|
||||
|
||||
#ifndef THREAD_LOCAL
|
||||
#define THREAD_LOCAL
|
||||
#endif
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Software floating-point underflow tininess-detection mode.
|
||||
*----------------------------------------------------------------------------*/
|
||||
extern THREAD_LOCAL uint_fast8_t softfloat_detectTininess;
|
||||
enum {
|
||||
softfloat_tininess_beforeRounding = 0,
|
||||
softfloat_tininess_afterRounding = 1
|
||||
};
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Software floating-point rounding mode. (Mode "odd" is supported only if
|
||||
| SoftFloat is compiled with macro 'SOFTFLOAT_ROUND_ODD' defined.)
|
||||
*----------------------------------------------------------------------------*/
|
||||
extern THREAD_LOCAL uint_fast8_t softfloat_roundingMode;
|
||||
enum {
|
||||
softfloat_round_near_even = 0,
|
||||
softfloat_round_minMag = 1,
|
||||
softfloat_round_min = 2,
|
||||
softfloat_round_max = 3,
|
||||
softfloat_round_near_maxMag = 4,
|
||||
softfloat_round_odd = 6
|
||||
};
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Software floating-point exception flags.
|
||||
*----------------------------------------------------------------------------*/
|
||||
extern THREAD_LOCAL uint_fast8_t softfloat_exceptionFlags;
|
||||
enum {
|
||||
softfloat_flag_inexact = 1,
|
||||
softfloat_flag_underflow = 2,
|
||||
softfloat_flag_overflow = 4,
|
||||
softfloat_flag_infinite = 8,
|
||||
softfloat_flag_invalid = 16
|
||||
};
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Routine to raise any or all of the software floating-point exception flags.
|
||||
*----------------------------------------------------------------------------*/
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void softfloat_raiseFlags( uint_fast8_t );
|
||||
void softfloat_raiseFlags( struct softfloat_state *, uint_fast8_t );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Integer-to-floating-point conversion routines.
|
||||
@@ -187,7 +148,7 @@ float16_t f32_to_f16( float32_t );
|
||||
float64_t f32_to_f64( float32_t );
|
||||
#ifdef SOFTFLOAT_FAST_INT64
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t f32_to_extF80( float32_t );
|
||||
extFloat80_t f32_to_extF80( struct softfloat_state *, float32_t );
|
||||
float128_t f32_to_f128( float32_t );
|
||||
#endif
|
||||
void f32_to_extF80M( float32_t, extFloat80_t * );
|
||||
@@ -223,7 +184,7 @@ float16_t f64_to_f16( float64_t );
|
||||
float32_t f64_to_f32( float64_t );
|
||||
#ifdef SOFTFLOAT_FAST_INT64
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t f64_to_extF80( float64_t );
|
||||
extFloat80_t f64_to_extF80( struct softfloat_state *, float64_t );
|
||||
float128_t f64_to_f128( float64_t );
|
||||
#endif
|
||||
void f64_to_extF80M( float64_t, extFloat80_t * );
|
||||
@@ -244,53 +205,47 @@ bool f64_le_quiet( float64_t, float64_t );
|
||||
bool f64_lt_quiet( float64_t, float64_t );
|
||||
bool f64_isSignalingNaN( float64_t );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Rounding precision for 80-bit extended double-precision floating-point.
|
||||
| Valid values are 32, 64, and 80.
|
||||
*----------------------------------------------------------------------------*/
|
||||
extern THREAD_LOCAL uint_fast8_t extF80_roundingPrecision;
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| 80-bit extended double-precision floating-point operations.
|
||||
*----------------------------------------------------------------------------*/
|
||||
#ifdef SOFTFLOAT_FAST_INT64
|
||||
uint_fast32_t extF80_to_ui32( extFloat80_t, uint_fast8_t, bool );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
uint_fast64_t extF80_to_ui64( extFloat80_t, uint_fast8_t, bool );
|
||||
uint_fast64_t extF80_to_ui64( struct softfloat_state *, extFloat80_t, uint_fast8_t, bool );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast32_t extF80_to_i32( extFloat80_t, uint_fast8_t, bool );
|
||||
int_fast32_t extF80_to_i32( struct softfloat_state *, extFloat80_t, uint_fast8_t, bool );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
int_fast64_t extF80_to_i64( extFloat80_t, uint_fast8_t, bool );
|
||||
int_fast64_t extF80_to_i64( struct softfloat_state *, extFloat80_t, uint_fast8_t, bool );
|
||||
uint_fast32_t extF80_to_ui32_r_minMag( extFloat80_t, bool );
|
||||
uint_fast64_t extF80_to_ui64_r_minMag( extFloat80_t, bool );
|
||||
int_fast32_t extF80_to_i32_r_minMag( extFloat80_t, bool );
|
||||
int_fast64_t extF80_to_i64_r_minMag( extFloat80_t, bool );
|
||||
float16_t extF80_to_f16( extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float32_t extF80_to_f32( extFloat80_t );
|
||||
float32_t extF80_to_f32( struct softfloat_state *, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float64_t extF80_to_f64( extFloat80_t );
|
||||
float64_t extF80_to_f64( struct softfloat_state *, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
float128_t extF80_to_f128( extFloat80_t );
|
||||
float128_t extF80_to_f128( struct softfloat_state *, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_roundToInt( extFloat80_t, uint_fast8_t, bool );
|
||||
extFloat80_t extF80_roundToInt( struct softfloat_state *, extFloat80_t, uint_fast8_t, bool );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_add( extFloat80_t, extFloat80_t );
|
||||
extFloat80_t extF80_add( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_sub( extFloat80_t, extFloat80_t );
|
||||
extFloat80_t extF80_sub( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_mul( extFloat80_t, extFloat80_t );
|
||||
extFloat80_t extF80_mul( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_div( extFloat80_t, extFloat80_t );
|
||||
extFloat80_t extF80_div( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_rem( extFloat80_t, extFloat80_t );
|
||||
extFloat80_t extF80_rem( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t extF80_sqrt( extFloat80_t );
|
||||
extFloat80_t extF80_sqrt( struct softfloat_state *, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
bool extF80_eq( extFloat80_t, extFloat80_t );
|
||||
bool extF80_eq( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
bool extF80_le( extFloat80_t, extFloat80_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
bool extF80_lt( extFloat80_t, extFloat80_t );
|
||||
bool extF80_lt( struct softfloat_state *, extFloat80_t, extFloat80_t );
|
||||
bool extF80_eq_signaling( extFloat80_t, extFloat80_t );
|
||||
bool extF80_le_quiet( extFloat80_t, extFloat80_t );
|
||||
bool extF80_lt_quiet( extFloat80_t, extFloat80_t );
|
||||
@@ -341,7 +296,7 @@ float16_t f128_to_f16( float128_t );
|
||||
float32_t f128_to_f32( float128_t );
|
||||
float64_t f128_to_f64( float128_t );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
extFloat80_t f128_to_extF80( float128_t );
|
||||
extFloat80_t f128_to_extF80( struct softfloat_state *, float128_t );
|
||||
float128_t f128_roundToInt( float128_t, uint_fast8_t, bool );
|
||||
float128_t f128_add( float128_t, float128_t );
|
||||
float128_t f128_sub( float128_t, float128_t );
|
||||
|
||||
@@ -44,10 +44,10 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
| should be simply `softfloat_exceptionFlags |= flags;'.
|
||||
*----------------------------------------------------------------------------*/
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void softfloat_raiseFlags( uint_fast8_t flags )
|
||||
void softfloat_raiseFlags( struct softfloat_state *state, uint_fast8_t flags )
|
||||
{
|
||||
|
||||
softfloat_exceptionFlags |= flags;
|
||||
state->exceptionFlags |= flags;
|
||||
|
||||
}
|
||||
|
||||
@@ -1,52 +0,0 @@
|
||||
|
||||
/*============================================================================
|
||||
|
||||
This C source file is part of the SoftFloat IEEE Floating-Point Arithmetic
|
||||
Package, Release 3e, by John R. Hauser.
|
||||
|
||||
Copyright 2011, 2012, 2013, 2014, 2015, 2016 The Regents of the University of
|
||||
California. All Rights Reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions, and the following disclaimer.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions, and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
3. Neither the name of the University nor the names of its contributors may
|
||||
be used to endorse or promote products derived from this software without
|
||||
specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS "AS IS", AND ANY
|
||||
EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE, ARE
|
||||
DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE FOR ANY
|
||||
DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
|
||||
ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
=============================================================================*/
|
||||
|
||||
#include <stdint.h>
|
||||
#include "platform.h"
|
||||
#include "internals.h"
|
||||
#include "specialize.h"
|
||||
#include "softfloat.h"
|
||||
|
||||
#ifndef THREAD_LOCAL
|
||||
#define THREAD_LOCAL
|
||||
#endif
|
||||
|
||||
THREAD_LOCAL uint_fast8_t softfloat_roundingMode = softfloat_round_near_even;
|
||||
THREAD_LOCAL uint_fast8_t softfloat_detectTininess = init_detectTininess;
|
||||
THREAD_LOCAL uint_fast8_t softfloat_exceptionFlags = 0;
|
||||
|
||||
THREAD_LOCAL uint_fast8_t extF80_roundingPrecision = 80;
|
||||
|
||||
@@ -77,5 +77,50 @@ struct extFloat80M { uint16_t signExp; uint64_t signif; };
|
||||
*----------------------------------------------------------------------------*/
|
||||
typedef struct extFloat80M extFloat80_t;
|
||||
|
||||
enum {
|
||||
softfloat_tininess_beforeRounding = 0,
|
||||
softfloat_tininess_afterRounding = 1
|
||||
};
|
||||
|
||||
enum {
|
||||
softfloat_round_near_even = 0,
|
||||
softfloat_round_minMag = 1,
|
||||
softfloat_round_min = 2,
|
||||
softfloat_round_max = 3,
|
||||
softfloat_round_near_maxMag = 4,
|
||||
softfloat_round_odd = 6
|
||||
};
|
||||
|
||||
enum {
|
||||
softfloat_flag_inexact = 1,
|
||||
softfloat_flag_underflow = 2,
|
||||
softfloat_flag_overflow = 4,
|
||||
softfloat_flag_infinite = 8,
|
||||
softfloat_flag_invalid = 16
|
||||
};
|
||||
|
||||
struct softfloat_state {
|
||||
/*----------------------------------------------------------------------------
|
||||
| Software floating-point underflow tininess-detection mode.
|
||||
*----------------------------------------------------------------------------*/
|
||||
uint8_t detectTininess; /* = init_detectTininess */
|
||||
/*----------------------------------------------------------------------------
|
||||
| Software floating-point rounding mode. (Mode "odd" is supported only if
|
||||
| SoftFloat is compiled with macro 'SOFTFLOAT_ROUND_ODD' defined.)
|
||||
*----------------------------------------------------------------------------*/
|
||||
uint8_t roundingMode; /* = softfloat_round_near_even */
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Software floating-point exception flags.
|
||||
*----------------------------------------------------------------------------*/
|
||||
uint8_t exceptionFlags; /* = 0 */
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Rounding precision for 80-bit extended double-precision floating-point.
|
||||
| Valid values are 32, 64, and 80.
|
||||
*----------------------------------------------------------------------------*/
|
||||
uint8_t roundingPrecision; /* = 80 */
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
@@ -136,7 +136,7 @@ uint_fast16_t
|
||||
| exception is raised.
|
||||
*----------------------------------------------------------------------------*/
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void softfloat_f32UIToCommonNaN( uint_fast32_t uiA, struct commonNaN *zPtr );
|
||||
void softfloat_f32UIToCommonNaN( struct softfloat_state *, uint_fast32_t uiA, struct commonNaN *zPtr );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Converts the common NaN pointed to by 'aPtr' into a 32-bit floating-point
|
||||
@@ -173,7 +173,7 @@ uint_fast32_t
|
||||
| exception is raised.
|
||||
*----------------------------------------------------------------------------*/
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void softfloat_f64UIToCommonNaN( uint_fast64_t uiA, struct commonNaN *zPtr );
|
||||
void softfloat_f64UIToCommonNaN( struct softfloat_state *, uint_fast64_t uiA, struct commonNaN *zPtr );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Converts the common NaN pointed to by 'aPtr' into a 64-bit floating-point
|
||||
@@ -222,7 +222,7 @@ uint_fast64_t
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void
|
||||
softfloat_extF80UIToCommonNaN(
|
||||
uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
|
||||
struct softfloat_state *, uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Converts the common NaN pointed to by 'aPtr' into an 80-bit extended
|
||||
@@ -244,6 +244,7 @@ struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr );
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
struct uint128
|
||||
softfloat_propagateNaNExtF80UI(
|
||||
struct softfloat_state *,
|
||||
uint_fast16_t uiA64,
|
||||
uint_fast64_t uiA0,
|
||||
uint_fast16_t uiB64,
|
||||
@@ -274,7 +275,7 @@ struct uint128
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
void
|
||||
softfloat_f128UIToCommonNaN(
|
||||
uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
|
||||
struct softfloat_state *, uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
|
||||
|
||||
/*----------------------------------------------------------------------------
|
||||
| Converts the common NaN pointed to by 'aPtr' into a 128-bit floating-point
|
||||
|
||||
@@ -63,7 +63,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
}
|
||||
|
||||
// Ops
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FADD(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FADD(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -79,11 +79,11 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_add(lhs, rhs);
|
||||
return extF80_add(state, lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSUB(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSUB(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -99,11 +99,11 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sub(lhs, rhs);
|
||||
return extF80_sub(state, lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FMUL(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FMUL(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -119,11 +119,11 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_mul(lhs, rhs);
|
||||
return extF80_mul(state, lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FDIV(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FDIV(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -139,11 +139,11 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_div(lhs, rhs);
|
||||
return extF80_div(state, lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -160,11 +160,35 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
/*
|
||||
* FPREM is not an IEEE-754 remainder. From the spec:
|
||||
*
|
||||
* Computes the remainder obtained from dividing the value in the ST(0)
|
||||
* register (the dividend) by the value in the ST(1) register (the divisor
|
||||
* or modulus), and stores the result in ST(0). The remainder represents the
|
||||
* following value:
|
||||
*
|
||||
* Remainder := ST(0) − (Q * ST(1))
|
||||
*
|
||||
* Here, Q is an integer value that is obtained by truncating the
|
||||
* floating-point number quotient of [ST(0) / ST(1)] toward zero.
|
||||
*
|
||||
* We implement this sequence literally. softfloat_round_minMag means
|
||||
* "truncate towards zero".
|
||||
*/
|
||||
extFloat80_t quotient = extF80_div(state, lhs, rhs);
|
||||
extFloat80_t Q = extF80_roundToInt(state, quotient, softfloat_round_minMag, true);
|
||||
bool Q_zero = Q.signif == 0 && (Q.signExp & ~(1 << 15)) == 0;
|
||||
|
||||
if (Q_zero) {
|
||||
return lhs;
|
||||
} else {
|
||||
return extF80_sub(state, lhs, extF80_mul(state, Q, rhs));
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM1(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM1(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -181,16 +205,16 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
return extF80_rem(state, lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs) {
|
||||
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(softfloat_state* state, const X80SoftFloat& lhs) {
|
||||
return extF80_roundToInt(state, lhs, state->roundingMode, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs, uint_fast8_t RoundMode) {
|
||||
return extF80_roundToInt(lhs, RoundMode, false);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(softfloat_state* state, const X80SoftFloat& lhs, uint_fast8_t RoundMode) {
|
||||
return extF80_roundToInt(state, lhs, RoundMode, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_SIG(const X80SoftFloat& lhs) {
|
||||
@@ -237,13 +261,14 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static void FCMP(const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
|
||||
*eq = extF80_eq(lhs, rhs);
|
||||
*lt = extF80_lt(lhs, rhs);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static void
|
||||
FCMP(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
|
||||
*eq = extF80_eq(state, lhs, rhs);
|
||||
*lt = extF80_lt(state, lhs, rhs);
|
||||
*nan = IsNan(lhs) || IsNan(rhs);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -261,16 +286,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
|
||||
LIBRARY_PRECISION Src2_d = Int;
|
||||
extFloat80_t Zero {0, 0};
|
||||
if (extF80_eq(state, lhs, Zero)) {
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Int = FRNDINT(state, rhs, softfloat_round_minMag);
|
||||
LIBRARY_PRECISION Src2_d = Int.ToFMax(state);
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
X80SoftFloat Src2_X80(state, Src2_d);
|
||||
X80SoftFloat Result = extF80_mul(state, lhs, Src2_X80);
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat F2XM1(const X80SoftFloat& lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat F2XM1(softfloat_state* state, const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -286,14 +315,14 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src1_d = lhs.ToFMax(state);
|
||||
LIBRARY_PRECISION Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
return X80SoftFloat(state, Result);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FYL2X(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FYL2X(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -310,14 +339,14 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src2_d = rhs;
|
||||
LIBRARY_PRECISION Src1_d = lhs.ToFMax(state);
|
||||
LIBRARY_PRECISION Src2_d = rhs.ToFMax(state);
|
||||
LIBRARY_PRECISION Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
return X80SoftFloat(state, Tmp);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FATAN(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FATAN(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -334,14 +363,14 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src2_d = rhs;
|
||||
LIBRARY_PRECISION Src1_d = lhs.ToFMax(state);
|
||||
LIBRARY_PRECISION Src2_d = rhs.ToFMax(state);
|
||||
LIBRARY_PRECISION Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
return X80SoftFloat(state, Tmp);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FTAN(const X80SoftFloat& lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FTAN(softfloat_state* state, const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -358,13 +387,13 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
LIBRARY_PRECISION Src_d = lhs.ToFMax(state);
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
return X80SoftFloat(state, Src_d);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSIN(const X80SoftFloat& lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSIN(softfloat_state* state, const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -380,13 +409,13 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
LIBRARY_PRECISION Src_d = lhs.ToFMax(state);
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
return X80SoftFloat(state, Src_d);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FCOS(const X80SoftFloat& lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FCOS(softfloat_state* state, const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
@@ -402,13 +431,13 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
LIBRARY_PRECISION Src_d = lhs.ToFMax(state);
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
return X80SoftFloat(state, Src_d);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSQRT(const X80SoftFloat& lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSQRT(softfloat_state* state, const X80SoftFloat& lhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm(R"(
|
||||
@@ -423,62 +452,55 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sqrt(lhs);
|
||||
return extF80_sqrt(state, lhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
operator float() const {
|
||||
const float32_t Result = extF80_to_f32(*this);
|
||||
float ToF32(softfloat_state* state) const {
|
||||
const float32_t Result = extF80_to_f32(state, *this);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
}
|
||||
|
||||
operator double() const {
|
||||
const float64_t Result = extF80_to_f64(*this);
|
||||
double ToF64(softfloat_state* state) const {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
operator BIGFLOAT() const {
|
||||
LIBRARY_PRECISION ToFMax(softfloat_state* state) const {
|
||||
#ifdef _WIN32
|
||||
return ToF64(state);
|
||||
#else
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
const float128_t Result = extF80_to_f128(state, *this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result {};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
return result;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
operator int16_t() const {
|
||||
auto rv = extF80_to_i32(*this, softfloat_roundingMode, false);
|
||||
if (rv > INT16_MAX) {
|
||||
return INT16_MAX;
|
||||
} else if (rv < INT16_MIN) {
|
||||
int16_t ToI16(softfloat_state* state) const {
|
||||
auto rv = extF80_to_i32(state, *this, state->roundingMode, false);
|
||||
if (rv > INT16_MAX || rv < INT16_MIN) {
|
||||
///< Indefinite value for 16-bit conversions.
|
||||
return INT16_MIN;
|
||||
} else {
|
||||
return rv;
|
||||
}
|
||||
}
|
||||
|
||||
operator int32_t() const {
|
||||
return extF80_to_i32(*this, softfloat_roundingMode, false);
|
||||
int32_t ToI32(softfloat_state* state) const {
|
||||
return extF80_to_i32(state, *this, state->roundingMode, false);
|
||||
}
|
||||
|
||||
operator int64_t() const {
|
||||
return extF80_to_i64(*this, softfloat_roundingMode, false);
|
||||
int64_t ToI64(softfloat_state* state) const {
|
||||
return extF80_to_i64(state, *this, state->roundingMode, false);
|
||||
}
|
||||
|
||||
operator uint64_t() const {
|
||||
return extF80_to_ui64(*this, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
void operator=(const float rhs) {
|
||||
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
void operator=(const double rhs) {
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
uint64_t ToUI64(softfloat_state* state) const {
|
||||
return extF80_to_ui64(state, *this, state->roundingMode, false);
|
||||
}
|
||||
|
||||
void operator=(const int16_t rhs) {
|
||||
@@ -509,18 +531,18 @@ struct FEX_PACKED X80SoftFloat {
|
||||
Sign = rhs.signExp >> 15;
|
||||
}
|
||||
|
||||
X80SoftFloat(const float rhs) {
|
||||
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
|
||||
X80SoftFloat(softfloat_state* state, const float rhs) {
|
||||
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(const double rhs) {
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
X80SoftFloat(softfloat_state* state, const double rhs) {
|
||||
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
X80SoftFloat(BIGFLOAT rhs) {
|
||||
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
|
||||
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
|
||||
#else
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
#endif
|
||||
|
||||
@@ -43,7 +43,8 @@ namespace DefaultValues {
|
||||
} // namespace DefaultValues
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
PATH_DATA_DIR_GLOBAL,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
@@ -53,8 +54,8 @@ enum Paths {
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
void SetDataDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_DATA_DIR_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
@@ -73,15 +74,15 @@ const fextl::string& GetTelemetryDirectory() {
|
||||
Path = TelemetryDirectory;
|
||||
Path += "/";
|
||||
} else {
|
||||
Path = Config::GetDataDirectory() + "Telemetry/";
|
||||
Path = Config::GetDataDirectory(false) + "Telemetry/";
|
||||
}
|
||||
}
|
||||
|
||||
return Path;
|
||||
}
|
||||
|
||||
const fextl::string& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
const fextl::string& GetDataDirectory(bool Global) {
|
||||
return Paths[PATH_DATA_DIR_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetConfigDirectory(bool Global) {
|
||||
@@ -230,27 +231,26 @@ void Load() {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, fextl::string PathName) {
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
|
||||
// Expand home if it exists
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
PathName.replace(0, 1, Home);
|
||||
return PathName;
|
||||
if (PathName.starts_with("~/")) {
|
||||
Home.append(PathName.begin() + 1, PathName.end());
|
||||
return Home;
|
||||
}
|
||||
|
||||
// Expand relative path to absolute
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char* RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
if (RealPath && FHU::Filesystem::Exists(RealPath)) {
|
||||
return RealPath;
|
||||
}
|
||||
|
||||
// Only return if it exists
|
||||
@@ -318,81 +318,61 @@ fextl::string FindContainerPrefix() {
|
||||
void ReloadMetaLayer() {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 0;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
}
|
||||
|
||||
fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
const fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, const fextl::string& PathName) {
|
||||
const auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_ROOTFS);
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
fextl::string NamedRootFS = GetDataDirectory(false) + "RootFS/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, *PathName);
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, *PathName);
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_THUNKCONFIG);
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
fextl::string NamedConfig = GetDataDirectory(false) + "ThunkConfigs/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
|
||||
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
|
||||
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_OUTPUTLOG);
|
||||
if (*PathName != "stdout" && *PathName != "stderr" && *PathName != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, *PathName);
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) && !FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
FEX_CONFIG_OPT(PathName, DUMPIR);
|
||||
if (PathName() != "no") {
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_DUMPIR);
|
||||
if (*PathName != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
|
||||
@@ -1,19 +1,6 @@
|
||||
{
|
||||
"Options": {
|
||||
"CPU": {
|
||||
"Core": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
|
||||
"TextDefault": "irjit",
|
||||
"ShortArg": "c",
|
||||
"Choices": [ "irjit", "host" ],
|
||||
"ArgumentHandler": "CoreHandler",
|
||||
"Desc": [
|
||||
"Which CPU core to use",
|
||||
"host only exists on x86_64",
|
||||
"[irjit, host]"
|
||||
]
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -76,6 +63,8 @@
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
"DISABLERPRES": "disablerpres",
|
||||
"ENABLESVEBITPERM": "enablesvebitperm",
|
||||
"DISABLESVEBITPERM": "disablesvebitperm",
|
||||
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
|
||||
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
|
||||
},
|
||||
@@ -97,6 +86,7 @@
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
|
||||
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
|
||||
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
|
||||
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
@@ -139,7 +129,7 @@
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks/",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
@@ -155,7 +145,7 @@
|
||||
},
|
||||
"ThunkHostLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks_32/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit host-side thunking libraries."
|
||||
]
|
||||
@@ -419,6 +409,14 @@
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
]
|
||||
},
|
||||
"StrictInProcessSplitLocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -506,6 +504,13 @@
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
},
|
||||
"NeedsSeccomp": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -6,7 +6,9 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
@@ -18,8 +20,8 @@ void InitializeStaticTables(OperatingMode Mode) {
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
@@ -38,14 +40,6 @@ void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThrea
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator* _SignalDelegation) {
|
||||
SignalDelegation = _SignalDelegation;
|
||||
}
|
||||
@@ -55,6 +49,10 @@ void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandl
|
||||
SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetThunkHandler(FEXCore::ThunkHandler* Handler) {
|
||||
ThunkHandler = Handler;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
@@ -26,14 +26,8 @@
|
||||
#include <stdint.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <queue>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
@@ -65,18 +59,21 @@ namespace Validation {
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
enum CoreRunningMode {
|
||||
MODE_RUN = 0,
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct ExitFunctionLinkData {
|
||||
uint64_t HostBranch;
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
struct CustomIRResult {
|
||||
void* Creator;
|
||||
void* Data;
|
||||
|
||||
CustomIRResult(void* Creator, void* Data)
|
||||
: Creator(Creator)
|
||||
, Data(Data) {}
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE = 128;
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
@@ -94,10 +91,6 @@ public:
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
@@ -136,7 +129,7 @@ public:
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
@@ -154,6 +147,7 @@ public:
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator* SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) override;
|
||||
void SetThunkHandler(FEXCore::ThunkHandler* Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
@@ -193,9 +187,10 @@ public:
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
std::optional<CustomIRResult>
|
||||
AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
|
||||
void AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) override;
|
||||
void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
@@ -206,8 +201,8 @@ public:
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck {false};
|
||||
@@ -218,17 +213,15 @@ public:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
@@ -239,32 +232,29 @@ public:
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
} Config;
|
||||
|
||||
|
||||
std::atomic_bool CoreShuttingDown {false};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
// CPUID depends on HostFeatures so needs to be initialized after that.
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl();
|
||||
ContextImpl(const FEXCore::HostFeatures& Features);
|
||||
~ContextImpl();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
@@ -283,9 +273,6 @@ public:
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LOGMAN_THROW_A_FMT(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}",
|
||||
Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
@@ -331,16 +318,6 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
@@ -349,27 +326,21 @@ public:
|
||||
return AtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
// If atomic-based TSO emulation is enabled for vector operations.
|
||||
bool IsVectorAtomicTSOEnabled() const {
|
||||
return VectorAtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
// If atomic-based TSO emulation is enabled for memcpy operations.
|
||||
bool IsMemcpyAtomicTSOEnabled() const {
|
||||
return MemcpyAtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
// Returns if Software TSO emulation is required.
|
||||
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
|
||||
// This will still return true if on a single thread and TSO is currently disabled.
|
||||
//
|
||||
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
|
||||
// we return consistent results.
|
||||
//
|
||||
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
|
||||
bool SoftwareTSORequired() const {
|
||||
if (SupportsHardwareTSO) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return Config.TSOEnabled;
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override {
|
||||
ExitOnHLT = true;
|
||||
}
|
||||
@@ -378,16 +349,20 @@ public:
|
||||
return ExitOnHLT;
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
|
||||
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
|
||||
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -410,6 +385,9 @@ private:
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool VectorAtomicTSOEmulationEnabled = false;
|
||||
bool MemcpyAtomicTSOEmulationEnabled = false;
|
||||
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -14,10 +13,12 @@
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
#include <cpu-features.h>
|
||||
#include <utils-vixl.h>
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <tuple>
|
||||
@@ -27,11 +28,12 @@ namespace FEXCore::CPU {
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
// Note that rax/rdx are rearranged here so we can coalesce cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
@@ -192,12 +194,12 @@ namespace x64 {
|
||||
} // namespace x64
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
// All but x19 and x29 are caller saved. eax/edx rearranged for cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 10> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
@@ -349,8 +351,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
}
|
||||
#endif
|
||||
|
||||
CPU.SetUp();
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
StaticRegisters = x64::SRA;
|
||||
@@ -373,6 +373,20 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::X86State::X86Reg Arm64Emitter::GetX86RegRelationToARMReg(ARMEmitter::Register Reg) {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
const auto& RegI = StaticRegisters[i];
|
||||
if (RegI == Reg) {
|
||||
// X86 Registers are mapped linerally from the StaticRegisters span.
|
||||
// Directly correlating Enum index to span index.
|
||||
return static_cast<FEXCore::X86State::X86Reg>(FEXCore::ToUnderlying(FEXCore::X86State::X86Reg::REG_RAX) + i);
|
||||
}
|
||||
}
|
||||
|
||||
// Unmapped register.
|
||||
return FEXCore::X86State::X86Reg::REG_INVALID;
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
@@ -421,7 +435,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
if (RequiredMoveSegments > 1) {
|
||||
// Only try to use this path if the number of segments is > 1.
|
||||
// `movz` is better than `orr` since hardware will rename or merge if possible when `movz` is used.
|
||||
const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Constant, RegSizeInBits(s));
|
||||
const auto IsImm = ARMEmitter::Emitter::IsImmLogical(Constant, RegSizeInBits(s));
|
||||
if (IsImm) {
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
if (NOPPad) {
|
||||
@@ -458,7 +472,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
|
||||
if (RequiredMoveSegments > 1 && ARMEmitter::Emitter::IsInt32(AlignedOffset)) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
@@ -466,7 +480,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
|
||||
// 21-bit signed integer here
|
||||
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
|
||||
if (vixl::IsInt21(SmallOffset)) {
|
||||
if (ARMEmitter::Emitter::IsInt21(SmallOffset)) {
|
||||
adr(Reg, SmallOffset);
|
||||
} else {
|
||||
// Need to use ADRP + ADD
|
||||
@@ -570,12 +584,53 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// Enable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
|
||||
if (SetFIZ) {
|
||||
// Insert MXCSR.DAZ in to FIZ
|
||||
ldr(TmpReg2.W(), STATE.R(), offsetof(FEXCore::Core::CPUState, mxcsr));
|
||||
bfxil(ARMEmitter::Size::i64Bit, TmpReg, TmpReg2, 6, 1);
|
||||
}
|
||||
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (SetPredRegs) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
//
|
||||
// Disable FPCR.NEP and FPCR.AH
|
||||
// Disable FPCR.NEP and FPCR.AH and FPCR.FIZ
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
|
||||
@@ -585,7 +640,8 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
(1U << 1) | // AH
|
||||
(1U << 0)); // FIZ
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
@@ -663,37 +719,37 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
ARMEmitter::Register TmpReg = ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
[[maybe_unused]] bool FoundRegister {};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & GPRFillMask)) {
|
||||
TmpReg = Reg;
|
||||
FoundRegister = true;
|
||||
break;
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask, std::optional<ARMEmitter::Register> OptionalReg,
|
||||
std::optional<ARMEmitter::Register> OptionalReg2) {
|
||||
auto FindTempReg = [this](uint32_t* GPRFillMask) -> std::optional<ARMEmitter::Register> {
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & *GPRFillMask)) {
|
||||
*GPRFillMask &= ~(1U << Reg.Idx());
|
||||
return std::make_optional(Reg);
|
||||
}
|
||||
}
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 2 GPRs for a temp");
|
||||
uint32_t TempGPRFillMask = GPRFillMask;
|
||||
if (!OptionalReg.has_value()) {
|
||||
OptionalReg = FindTempReg(&TempGPRFillMask);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(FoundRegister, "Didn't have an SRA register to use as a temporary while spilling!");
|
||||
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// Enable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
if (!OptionalReg2.has_value()) {
|
||||
OptionalReg2 = FindTempReg(&TempGPRFillMask);
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(OptionalReg.has_value() && OptionalReg2.has_value(), "Didn't have an SRA register to use as a temporary while "
|
||||
"spilling!");
|
||||
|
||||
auto TmpReg = *OptionalReg;
|
||||
auto TmpReg2 = *OptionalReg2;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Load STATE in from the CPU area as x28 is not callee saved in the ARM64EC ABI.
|
||||
ldr(TmpReg.X(), ARMEmitter::Reg::r18, TEB_CPU_AREA_OFFSET);
|
||||
ldr(STATE, TmpReg, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
@@ -704,19 +760,9 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
|
||||
FillSpecialRegs(TmpReg, TmpReg2, true, FPRs);
|
||||
|
||||
if (FPRs) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
@@ -4,11 +4,6 @@
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#endif
|
||||
@@ -17,6 +12,7 @@
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
@@ -70,6 +66,14 @@ constexpr auto VTMP2 = ARMEmitter::VReg::v17;
|
||||
// Entry/Exit ABI
|
||||
constexpr auto EC_CALL_CHECKER_PC_REG = ARMEmitter::XReg::x9;
|
||||
constexpr auto EC_ENTRY_CPUAREA_REG = ARMEmitter::XReg::x17;
|
||||
|
||||
// These structures are not included in the standard Windows headers, define the offsets of members we care about for EC here.
|
||||
constexpr size_t TEB_CPU_AREA_OFFSET = 0x1788;
|
||||
constexpr size_t TEB_PEB_OFFSET = 0x60;
|
||||
constexpr size_t PEB_EC_CODE_BITMAP_OFFSET = 0x368;
|
||||
constexpr size_t CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET = 0x1;
|
||||
constexpr size_t CPU_AREA_EMULATOR_STACK_BASE_OFFSET = 0x8;
|
||||
constexpr size_t CPU_AREA_EMULATOR_DATA_OFFSET = 0x30;
|
||||
#endif
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
@@ -86,7 +90,6 @@ protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
@@ -97,12 +100,19 @@ protected:
|
||||
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
// Returning REG_INVALID if there was no mapping.
|
||||
FEXCore::X86State::X86Reg GetX86RegRelationToARMReg(ARMEmitter::Register Reg);
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U,
|
||||
std::optional<ARMEmitter::Register> OptionalReg = std::nullopt,
|
||||
std::optional<ARMEmitter::Register> OptionalReg2 = std::nullopt);
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0110'0000'0000'0111'1111'1111'1111'1111U;
|
||||
|
||||
@@ -1,51 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore {
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
std::fstream Output;
|
||||
Output.open("output.csv", std::fstream::out | std::fstream::binary);
|
||||
|
||||
if (!Output.is_open()) {
|
||||
return;
|
||||
}
|
||||
|
||||
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
|
||||
|
||||
for (auto it : SamplingMap) {
|
||||
if (!it.second->TotalCalls) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Output << "0x" << std::hex << it.first << ", " << std::dec << it.second->Min << ", " << std::dec << it.second->Max << ", " << std::dec
|
||||
<< it.second->TotalTime << ", " << std::dec << it.second->TotalCalls << ", " << std::dec
|
||||
<< ((double)it.second->TotalTime / (double)it.second->TotalCalls) << std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData* BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
auto it = SamplingMap.find(RIP);
|
||||
if (it != SamplingMap.end()) {
|
||||
return it->second;
|
||||
}
|
||||
BlockData* NewData = new BlockData {};
|
||||
memset(NewData, 0, sizeof(BlockData));
|
||||
NewData->Min = ~0ULL;
|
||||
SamplingMap[RIP] = NewData;
|
||||
return NewData;
|
||||
}
|
||||
|
||||
BlockSamplingData::~BlockSamplingData() {
|
||||
DumpBlockData();
|
||||
for (auto it : SamplingMap) {
|
||||
delete it.second;
|
||||
}
|
||||
SamplingMap.clear();
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -1,25 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockSamplingData {
|
||||
public:
|
||||
struct BlockData {
|
||||
uint64_t Start, End;
|
||||
uint64_t Min, Max;
|
||||
uint64_t TotalTime;
|
||||
uint64_t TotalCalls;
|
||||
};
|
||||
|
||||
BlockData* GetBlockData(uint64_t RIP);
|
||||
~BlockSamplingData();
|
||||
|
||||
void DumpBlockData();
|
||||
|
||||
private:
|
||||
std::unordered_map<uint64_t, BlockData*> SamplingMap;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -34,11 +34,6 @@ namespace CodeSerialize {
|
||||
}
|
||||
|
||||
namespace CPU {
|
||||
struct CPUBackendFeatures {
|
||||
bool SupportsFlags = false;
|
||||
bool SupportsVTBL2 = false;
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
@@ -53,11 +48,6 @@ namespace CPU {
|
||||
CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
/**
|
||||
* @return The name of this backend
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual fextl::string GetName() = 0;
|
||||
|
||||
struct CompiledCode {
|
||||
// Where this code block begins.
|
||||
@@ -129,9 +119,6 @@ namespace CPU {
|
||||
*
|
||||
* This is a thread specific compilation unit since there is one CPUBackend per guest thread
|
||||
*
|
||||
* If NeedsOpDispatch is returning false then IR and DebugData may be null and the expectation is that the code will still compile
|
||||
* FEXCore::Core::ThreadState* is valid at the time of compilation.
|
||||
*
|
||||
* @param IR - IR that maps to the IR for this RIP
|
||||
* @param DebugData - Debug data that is available for this IR indirectly
|
||||
*
|
||||
@@ -154,26 +141,6 @@ namespace CPU {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
*
|
||||
* @return Currently unused
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual void* MapRegion(void* HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
|
||||
*
|
||||
* This is useful if the FEXCore Frontend hits an x86-64 instruction that isn't understood but can continue regardless
|
||||
*
|
||||
* This is useful for example, a VM based CPUbackend
|
||||
*
|
||||
* @return true if it needs the IR
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual bool NeedsOpDispatch() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
/**
|
||||
|
||||
@@ -34,6 +34,8 @@ namespace ProductNames {
|
||||
static const char ARM_A76AE[] = "Cortex-A76AE";
|
||||
static const char ARM_V1[] = "Neoverse V1";
|
||||
static const char ARM_V2[] = "Neoverse V2";
|
||||
static const char ARM_V3[] = "Neoverse V3";
|
||||
static const char ARM_V3AE[] = "Neoverse V3AE";
|
||||
static const char ARM_A77[] = "Cortex-A77";
|
||||
static const char ARM_A78[] = "Cortex-A78";
|
||||
static const char ARM_A78AE[] = "Cortex-A78AE";
|
||||
@@ -41,13 +43,16 @@ namespace ProductNames {
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_A725[] = "Cortex-A725";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_X925[] = "Cortex-X925";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_N3[] = "Neoverse N3";
|
||||
static const char ARM_E1[] = "Neoverse E1";
|
||||
static const char ARM_A35[] = "Cortex-A35";
|
||||
static const char ARM_A53[] = "Cortex-A53";
|
||||
@@ -67,8 +72,18 @@ namespace ProductNames {
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
|
||||
static const char ARM_Firestorm[] = "Apple Firestorm";
|
||||
static const char ARM_Icestorm[] = "Apple Icestorm";
|
||||
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
|
||||
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
|
||||
static const char ARM_Firestorm_M1Pro[] = "Apple Firestorm (M1 Pro)";
|
||||
static const char ARM_Icestorm_M1Pro[] = "Apple Icestorm (M1 Pro)";
|
||||
static const char ARM_Firestorm_M1Max[] = "Apple Firestorm (M1 Max)";
|
||||
static const char ARM_Icestorm_M1Max[] = "Apple Icestorm (M1 Max)";
|
||||
static const char ARM_Avalanche_M2[] = "Apple Avalanche (M2)";
|
||||
static const char ARM_Blizzard_M2[] = "Apple Blizzard (M2)";
|
||||
static const char ARM_Avalanche_M2Pro[] = "Apple Avalanche (M2 Pro)";
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
#else
|
||||
@@ -81,20 +96,39 @@ static uint32_t GetCPUID() {
|
||||
return CPU;
|
||||
}
|
||||
|
||||
struct CPUFamily {
|
||||
uint32_t Stepping : 4;
|
||||
uint32_t Model : 4;
|
||||
uint32_t ExtendedModel : 4;
|
||||
uint32_t FamilyID : 4;
|
||||
uint32_t ExtendedFamilyID : 8;
|
||||
uint32_t ProcessorType : 4;
|
||||
};
|
||||
|
||||
constexpr static uint32_t GenerateFamily(const CPUFamily Family) {
|
||||
return Family.Stepping | (Family.Model << 4) | (Family.FamilyID << 8) | (Family.ProcessorType << 12) | (Family.ExtendedModel << 16) |
|
||||
(Family.ExtendedFamilyID << 20);
|
||||
}
|
||||
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = 0 | // Stepping
|
||||
(0xA << 4) | // Model
|
||||
(0xF << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(0 << 16) | // Extended model ID
|
||||
(1 << 20); // Extended family ID
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
.Stepping = 0,
|
||||
.Model = 0xA,
|
||||
.ExtendedModel = 0,
|
||||
.FamilyID = 0xF,
|
||||
.ExtendedFamilyID = 1,
|
||||
.ProcessorType = 0,
|
||||
});
|
||||
|
||||
#else
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = 0 | // Stepping
|
||||
(0x7 << 4) | // Model
|
||||
(0x6 << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(1 << 16) | // Extended model ID
|
||||
(0x0 << 20); // Extended family ID
|
||||
constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
.Stepping = 1,
|
||||
.Model = 6,
|
||||
.ExtendedModel = 0xA,
|
||||
.FamilyID = 6,
|
||||
.ExtendedFamilyID = 0,
|
||||
.ProcessorType = 0,
|
||||
});
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
@@ -109,27 +143,16 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
uint64_t MIDR {};
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec {};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
std::array<char, 18> Data;
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFileToBuffer(MIDRPath, Data) == sizeof(Data)) {
|
||||
uint64_t NewMIDR {};
|
||||
std::string_view MIDRView(Data.data(), sizeof(Data));
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
auto NewMIDR = CTX->HostFeatures.CPUMIDRs[i];
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
|
||||
struct CPUMIDR {
|
||||
@@ -142,12 +165,22 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 43> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
{0x61, 0x039, 1, ProductNames::ARM_Avalanche_M2Max}, // Apple Avalanche (M2 Max)
|
||||
{0x61, 0x035, 1, ProductNames::ARM_Avalanche_M2Pro}, // Apple Avalanche (M2 Pro)
|
||||
{0x61, 0x033, 1, ProductNames::ARM_Avalanche_M2}, // Apple Avalanche (M2)
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
@@ -183,7 +216,13 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd07, 1, ProductNames::ARM_A57}, // A57
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x61, 0x038, 0, ProductNames::ARM_Blizzard_M2Max}, // Apple Blizzard (M2 Max)
|
||||
{0x61, 0x034, 0, ProductNames::ARM_Blizzard_M2Pro}, // Apple Blizzard (M2 Pro)
|
||||
{0x61, 0x032, 0, ProductNames::ARM_Blizzard_M2}, // Apple Blizzard (M2)
|
||||
{0x61, 0x028, 0, ProductNames::ARM_Icestorm_M1Max}, // Apple Icestorm (M1 Max)
|
||||
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
|
||||
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
@@ -598,8 +637,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
// Disable Enhanced REP MOVS when TSO is enabled.
|
||||
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
|
||||
// This is due to LRCPC performance on Cortex being abysmal.
|
||||
// Only enable EnhancedREPMOVS if SoftwareTSO isn't required OR if MemcpySetTSO is not enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->SoftwareTSORequired() == false || MemcpySetTSOEnabled() == false;
|
||||
// Only enable EnhancedREPMOVS if atomic memcpy tso emulation isn't enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->IsMemcpyAtomicTSOEnabled() == false;
|
||||
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
|
||||
|
||||
// Number of subfunctions
|
||||
@@ -628,7 +667,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(1 << 23) | // CLFLUSHOPT instruction
|
||||
(CTX->HostFeatures.SupportsCLWB << 24) | // CLWB instruction
|
||||
(1 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
@@ -762,7 +801,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
if (FrequencyHz) {
|
||||
Res.eax = 1;
|
||||
Res.ebx = CTX->Config.SmallTSCScale() ? FEXCore::Context::TSC_SCALE : 1;
|
||||
Res.ebx = 1U << CTX->Config.TSCScale;
|
||||
Res.ecx = FrequencyHz;
|
||||
}
|
||||
return Res;
|
||||
@@ -1175,7 +1214,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {
|
||||
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
@@ -118,8 +118,6 @@ private:
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
@@ -20,7 +19,6 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -29,16 +27,17 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
@@ -75,12 +74,10 @@ $end_info$
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl()
|
||||
: CPUID {this}
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
: HostFeatures {Features}
|
||||
, CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
@@ -94,8 +91,13 @@ ContextImpl::ContextImpl()
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
if (FEXCore::GetCycleCounterFrequency() >= FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
Config.SmallTSCScale = false;
|
||||
uint64_t FrequencyCounter = FEXCore::GetCycleCounterFrequency();
|
||||
if (FrequencyCounter && FrequencyCounter < FEXCore::Context::TSC_SCALE_MAXIMUM && Config.SmallTSCScale()) {
|
||||
// Scale TSC until it is at the minimum required.
|
||||
while (FrequencyCounter < FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
FrequencyCounter <<= 1;
|
||||
++Config.TSCScale;
|
||||
}
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
@@ -190,6 +192,9 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC)) & 1;
|
||||
|
||||
// CF is inverted in our representation, undo the invert here.
|
||||
CF ^= 1;
|
||||
|
||||
// Pack in to EFLAGS
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_RAW_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_RAW_LOC;
|
||||
@@ -293,10 +298,10 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate packed NZCV
|
||||
// Calculate packed NZCV. Note CF is inverted.
|
||||
uint32_t Packed_NZCV {};
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 0 : 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC);
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC) : 0;
|
||||
memcpy(&Frame->State.flags[X86State::RFLAG_NZCV_LOC], &Packed_NZCV, sizeof(Packed_NZCV));
|
||||
@@ -309,20 +314,10 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
|
||||
|
||||
bool ContextImpl::InitCore() {
|
||||
// Initialize the CPU core signal handlers & DispatcherConfig
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT: BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures(); break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM:
|
||||
// Do nothing
|
||||
break;
|
||||
default: LogMan::Msg::EFmt("Unknown core configuration"); return false;
|
||||
}
|
||||
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.SupportsAVX = HostFeatures.SupportsAVX,
|
||||
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
|
||||
@@ -351,9 +346,8 @@ bool ContextImpl::InitCore() {
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
// WIN32 always needs the interrupt fault check to be enabled.
|
||||
#elif !defined(_M_ARM64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
@@ -373,14 +367,15 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
|
||||
FEXCore::Context::ExitReason ContextImpl::RunUntilExit(FEXCore::Core::InternalThreadState* Thread) {
|
||||
ExecutionThread(Thread);
|
||||
while (true) {
|
||||
auto reason = Thread->ExitReason;
|
||||
|
||||
// Don't return if a custom exit handling the exit
|
||||
if (!CustomExitHandler || reason == ExitReason::EXIT_SHUTDOWN) {
|
||||
return reason;
|
||||
}
|
||||
CoreShuttingDown.store(true);
|
||||
|
||||
if (CustomExitHandler) {
|
||||
CustomExitHandler(Thread, FEXCore::Context::ExitReason::EXIT_SHUTDOWN);
|
||||
return Thread->ExitReason;
|
||||
}
|
||||
|
||||
return FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
}
|
||||
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
@@ -390,12 +385,6 @@ void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
|
||||
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
@@ -421,20 +410,14 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM: Thread->CPUBackend = CustomCPUFactory(this, Thread); break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown core configuration"); break;
|
||||
}
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
@@ -446,7 +429,6 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::C
|
||||
}
|
||||
|
||||
// Set up the thread manager state
|
||||
Thread->ThreadManager.parent_tid = ParentTID;
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
InitializeCompiler(Thread);
|
||||
@@ -479,8 +461,14 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
StrictSplitLockMutex = 0;
|
||||
}
|
||||
} else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::unlock(&StrictSplitLockMutex);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
@@ -488,6 +476,9 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::lock(&StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -606,6 +597,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
const FEXCore::X86Tables::DecodedInst* DecodedInfo {nullptr};
|
||||
@@ -614,6 +610,20 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
// Do a partial register cache flush before every instruction. This
|
||||
// prevents cross-instruction static register caching, while allowing
|
||||
// context load/stores to be optimized within a block. Theoretically,
|
||||
// this flush is not required for correctness, all mandatory flushes are
|
||||
// included in instruction-specific handlers. Instead, this is a blunt
|
||||
// heuristic to make the register cache less aggressive, as the current
|
||||
// RA generates bad code in common cases with tied registers otherwise.
|
||||
//
|
||||
// However, it makes our exception handling behaviour more predictable.
|
||||
// It is potentially correctness bearing in that sense, but that is a
|
||||
// side effect here and (if that behaviour is required) we should handle
|
||||
// that more explicitly later.
|
||||
Thread->OpDispatcher->FlushRegisterCache(true);
|
||||
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
@@ -804,14 +814,6 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// If the target PC is EC code, mark it in the L2 and return straight to the dispatcher
|
||||
// so it can handle the call/return.
|
||||
if (Thread->LookupCache->CheckPageEC(GuestRIP)) {
|
||||
return GuestRIP;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
@@ -918,15 +920,6 @@ void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// If it is the parent thread that died then just leave
|
||||
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
|
||||
|
||||
if (Thread->ThreadManager.parent_tid == 0) {
|
||||
CoreShuttingDown.store(true);
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
|
||||
if (CustomExitHandler) {
|
||||
CustomExitHandler(Thread->ThreadManager.TID, Thread->ExitReason);
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
@@ -985,7 +978,8 @@ void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
std::optional<CustomIRResult>
|
||||
ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
@@ -995,10 +989,50 @@ CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIR
|
||||
|
||||
if (!InsertedIterator.second) {
|
||||
const auto& [fn, Creator, Data] = InsertedIterator.first->second;
|
||||
return CustomIRResult(std::move(lk), Creator, Data);
|
||||
} else {
|
||||
lk.unlock();
|
||||
return CustomIRResult(std::move(lk), 0, 0);
|
||||
return CustomIRResult(Creator, Data);
|
||||
}
|
||||
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) {
|
||||
LOGMAN_THROW_AA_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
if (!Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}", Entrypoint, GuestThunkEntrypoint);
|
||||
|
||||
auto Result = AddCustomIREntrypoint(
|
||||
Entrypoint,
|
||||
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
if (GPRSize == 8) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), X86State::REG_R11, IR::GPRClass, GPRSize);
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
ThunkHandler, (void*)GuestThunkEntrypoint);
|
||||
|
||||
if (Result.has_value()) {
|
||||
if (Result->Creator != ThunkHandler) {
|
||||
ERROR_AND_DIE_FMT("Input address for AddThunkTrampoline is already linked by another module");
|
||||
}
|
||||
if (Result->Data != (void*)GuestThunkEntrypoint) {
|
||||
// NOTE: This may happen in Vulkan thunks if the Vulkan driver resolves two different symbols
|
||||
// to the same function (e.g. vkGetPhysicalDeviceFeatures2/vkGetPhysicalDeviceFeatures2KHR)
|
||||
LogMan::Msg::EFmt("Input address for AddThunkTrampoline is already linked elsewhere");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1021,12 +1055,6 @@ void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) {
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
|
||||
@@ -62,14 +62,6 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
#ifdef _M_ARM_64EC
|
||||
// These structures are not included in the standard Windows headers, define them here
|
||||
static constexpr size_t TEBCPUAreaOffset = 0x1788;
|
||||
static constexpr size_t CPUAreaInSyscallCallbackOffset = 0x1;
|
||||
static constexpr size_t CPUAreaEmulatorStackLimitOffset = 0x8;
|
||||
static constexpr size_t CPUAreaEmulatorDataOffset = 0x30;
|
||||
ARMEmitter::SingleUseForwardLabel ExitEC;
|
||||
#endif
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
|
||||
// Push all the register we need to save
|
||||
@@ -94,7 +86,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorDataOffset);
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
FillStaticRegs();
|
||||
|
||||
// Enter JIT
|
||||
@@ -102,17 +94,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
|
||||
// Load ThreadState and write the target PC there
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorDataOffset);
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
str(EC_CALL_CHECKER_PC_REG, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Swap stacks to the emulator stack
|
||||
ldr(TMP1, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorStackLimitOffset);
|
||||
ldr(TMP1, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_STACK_BASE_OFFSET);
|
||||
add(ARMEmitter::Size::i64Bit, StaticRegisters[X86State::REG_RSP], ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, TMP1, 0);
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
FillSpecialRegs(TMP1, TMP2, false, true);
|
||||
|
||||
// Enter JIT
|
||||
#endif
|
||||
@@ -168,10 +158,6 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
#ifdef _M_ARM_64EC
|
||||
// The LSB of an L2 page entry indicates if this page contains EC code
|
||||
tbnz(TMP1, 0, &ExitEC);
|
||||
#endif
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
@@ -198,23 +184,13 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP3, TMP1);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
{
|
||||
Bind(&ExitEC);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
}
|
||||
#endif
|
||||
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -232,14 +208,16 @@ void Dispatcher::EmitDispatcher() {
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPUAreaInSyscallCallbackOffset);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
@@ -259,10 +237,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
strb(ARMEmitter::WReg::zr, TMP2, CPUAreaInSyscallCallbackOffset);
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
@@ -270,6 +249,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
|
||||
br(TMP1);
|
||||
}
|
||||
@@ -278,20 +258,43 @@ void Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::SingleUseForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
|
||||
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
tbz(TMP1, 0, &l_NotECCode);
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
#endif
|
||||
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, TMP3);
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPUAreaInSyscallCallbackOffset);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
@@ -309,10 +312,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
strb(ARMEmitter::WReg::zr, TMP1, CPUAreaInSyscallCallbackOffset);
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP1, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
@@ -320,6 +324,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
@@ -274,12 +274,10 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -557,12 +555,10 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -632,7 +628,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
|
||||
uint16_t map_select = 1;
|
||||
uint16_t pp = 0;
|
||||
const uint8_t Byte1 = ReadByte();
|
||||
@@ -678,7 +673,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &VEXTableOps[Op];
|
||||
|
||||
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 && LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
|
||||
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
@@ -696,12 +690,8 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
}
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
|
||||
/* uint8_t P1 = */ ReadByte();
|
||||
/* uint8_t P2 = */ ReadByte();
|
||||
/* uint8_t P3 = */ ReadByte();
|
||||
uint8_t EVEXOp = ReadByte();
|
||||
return NormalOp(&EVEXTableOps[EVEXOp], EVEXOp);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
|
||||
@@ -969,8 +959,14 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
// If the target RIP is within the symbol ranges then we are golden
|
||||
if (TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress) {
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
#endif
|
||||
|
||||
if (ValidMultiblockMember) {
|
||||
// Update our conditional branch ranges before we return
|
||||
if (Conditional) {
|
||||
MaxCondBranchForward = std::max(MaxCondBranchForward, TargetRIP);
|
||||
@@ -979,12 +975,12 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
BlocksToDecode.insert(FallthroughRIP);
|
||||
CurrentBlockTargets.insert(FallthroughRIP);
|
||||
}
|
||||
}
|
||||
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
BlocksToDecode.insert(TargetRIP);
|
||||
CurrentBlockTargets.insert(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
@@ -1081,6 +1077,8 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1117,7 +1115,6 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) [[unlikely]] {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
@@ -1125,6 +1122,14 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
DecodeInst->InstSize = 0;
|
||||
}
|
||||
|
||||
if (!ErrorDuringDecoding) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
auto TableInfo = DecodedBuffer[BlockStartOffset + BlockNumberOfInstructions].TableInfo;
|
||||
if (!TableInfo || !TableInfo->OpcodeDispatcher) {
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, RIPToDecode + PCOffset + DecodeInst->InstSize);
|
||||
++TotalInstructions;
|
||||
@@ -1132,7 +1137,17 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
++DecodedSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) {
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) [[unlikely]] {
|
||||
if (!EntryBlock) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
TotalInstructions -= CurrentBlockDecoding.NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
BlockNumberOfInstructions = 0;
|
||||
InstStream -= PCOffset;
|
||||
CurrentBlockTargets.clear();
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1162,6 +1177,9 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
InstStream += DecodeInst->InstSize;
|
||||
}
|
||||
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
CurrentBlockTargets.clear();
|
||||
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
HasBlocks.emplace(RIPToDecode);
|
||||
|
||||
@@ -1169,6 +1187,8 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockInfo.TotalInstructionCount += BlockNumberOfInstructions;
|
||||
|
||||
EntryBlock = false;
|
||||
}
|
||||
|
||||
for (auto CodePage : CodePages) {
|
||||
|
||||
@@ -104,6 +104,7 @@ private:
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
|
||||
DecodedBlockInformation BlockInfo;
|
||||
fextl::set<uint64_t> CurrentBlockTargets;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
@@ -120,7 +121,6 @@ private:
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -1,293 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#define XBYAK64
|
||||
#define XBYAK_NO_EXCEPTION
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
// Log2 of the blocksize in 32-bit words
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
[[maybe_unused]]
|
||||
static uint32_t GetDCZID() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], DCZID_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static uint32_t GetFPCR() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], FPCR" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static void SetFPCR(uint64_t Value) {
|
||||
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
|
||||
}
|
||||
|
||||
static uint32_t GetMIDR() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], MIDR_EL1" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetDCZID() {
|
||||
// Return unsupported
|
||||
return DCZID_DZP_MASK;
|
||||
}
|
||||
#endif
|
||||
|
||||
static void OverrideFeatures(HostFeatures* Features, uint64_t ForceSVEWidth) {
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(HostFeatures, HOSTFEATURES);
|
||||
if (!HostFeatures()) {
|
||||
// Early exit if no features are overriden.
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features->FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features->FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
#define GET_SINGLE_OPTION(name, enum_name) \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
|
||||
ENABLE_DISABLE_OPTION(SupportsSVE128, SVE, SVE);
|
||||
ENABLE_DISABLE_OPTION(SupportsAFP, AFP, AFP);
|
||||
ENABLE_DISABLE_OPTION(SupportsRCPC, LRCPC, LRCPC);
|
||||
ENABLE_DISABLE_OPTION(SupportsTSOImm9, LRCPC2, LRCPC2);
|
||||
ENABLE_DISABLE_OPTION(SupportsCSSC, CSSC, CSSC);
|
||||
ENABLE_DISABLE_OPTION(SupportsPMULL_128Bit, PMULL128, PMULL128);
|
||||
ENABLE_DISABLE_OPTION(SupportsRAND, RNG, RNG);
|
||||
ENABLE_DISABLE_OPTION(SupportsCLZERO, CLZERO, CLZERO);
|
||||
ENABLE_DISABLE_OPTION(SupportsAtomics, Atomics, ATOMICS);
|
||||
ENABLE_DISABLE_OPTION(SupportsFCMA, FCMA, FCMA);
|
||||
ENABLE_DISABLE_OPTION(SupportsFlagM, FlagM, FLAGM);
|
||||
ENABLE_DISABLE_OPTION(SupportsFlagM2, FlagM2, FLAGM2);
|
||||
ENABLE_DISABLE_OPTION(SupportsRPRES, RPRES, RPRES);
|
||||
ENABLE_DISABLE_OPTION(SupportsPreserveAllABI, PRESERVEALLABI, PRESERVEALLABI);
|
||||
GET_SINGLE_OPTION(Crypto, CRYPTO);
|
||||
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
#undef GET_SINGLE_OPTION
|
||||
|
||||
if (EnableCrypto) {
|
||||
Features->SupportsAES = true;
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsSHA = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
Features->SupportsAES256 = true;
|
||||
} else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsSHA = false;
|
||||
Features->SupportsPMULL_128Bit = false;
|
||||
Features->SupportsAES256 = false;
|
||||
}
|
||||
|
||||
///< Only force enable SVE256 if SVE is already enabled and ForceSVEWidth is set to >= 256.
|
||||
Features->SupportsSVE256 = ForceSVEWidth && ForceSVEWidth >= 256;
|
||||
}
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
// Vixl simulator doesn't support AFP.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kAFP);
|
||||
// Vixl simulator doesn't support RPRES.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kRPRES);
|
||||
#elif !defined(_WIN32)
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
// Need to use ID registers in WINE.
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
|
||||
FEX_CONFIG_OPT(ForceSVEWidth, FORCESVEWIDTH);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) && Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsAFP = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
SupportsCSSC = Features.Has(vixl::CPUFeatures::Feature::kCSSC);
|
||||
SupportsFCMA = Features.Has(vixl::CPUFeatures::Feature::kFcma);
|
||||
SupportsFlagM = Features.Has(vixl::CPUFeatures::Feature::kFlagM);
|
||||
SupportsFlagM2 = Features.Has(vixl::CPUFeatures::Feature::kAXFlag);
|
||||
SupportsRPRES = Features.Has(vixl::CPUFeatures::Feature::kRPRES);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
SupportsSVE128 = ForceSVEWidth() ? ForceSVEWidth() >= 128 : true;
|
||||
SupportsSVE256 = ForceSVEWidth() ? ForceSVEWidth() >= 256 : true;
|
||||
#else
|
||||
SupportsSVE128 = Features.Has(vixl::CPUFeatures::Feature::kSVE2);
|
||||
SupportsSVE256 = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
SupportsAVX = true;
|
||||
|
||||
SupportsAES256 = SupportsAVX && SupportsAES;
|
||||
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps = (1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
SetFPCR(FPCR);
|
||||
FPCR = GetFPCR();
|
||||
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
|
||||
if (SupportsRAND) {
|
||||
const auto MIDR = GetMIDR();
|
||||
constexpr uint32_t Implementer_QCOM = 0x51;
|
||||
constexpr uint32_t PartNum_Oryon1 = 0x001;
|
||||
const uint32_t MIDR_Implementer = (MIDR >> 24) & 0xFF;
|
||||
const uint32_t MIDR_PartNum = (MIDR >> 4) & 0xFFF;
|
||||
if (MIDR_Implementer == Implementer_QCOM && MIDR_PartNum == PartNum_Oryon1) {
|
||||
// Work around an errata in Qualcomm's Oryon.
|
||||
// While this CPU implements the RAND extension:
|
||||
// - The RNDR register works.
|
||||
// - The RNDRRS register will never read a random number. (Always return failure)
|
||||
// This is contrary to x86 RNG behaviour where it allows spurious failure with RDSEED, but guarantees eventual success.
|
||||
// This manifested itself on Linux when an x86 processor failed to guarantee forward progress and boot of services would infinite
|
||||
// loop. Just disable this extension if this CPU is detected.
|
||||
SupportsRAND = false;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
// Simulator doesn't support SHA
|
||||
SupportsSHA = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
uint32_t DCZID_Log2 = DCZID & DCZID_BS_MASK;
|
||||
uint32_t DCZID_Bytes = (1 << DCZID_Log2) * sizeof(uint32_t);
|
||||
// If the DC ZVA size matches the emulated cache line size
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(_M_X86_64)
|
||||
// Hardcoded cacheline size.
|
||||
DCacheLineSize = 64U;
|
||||
ICacheLineSize = 64U;
|
||||
|
||||
#if !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu X86Features {};
|
||||
SupportsAES = X86Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = X86Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = X86Features.has(Xbyak::util::Cpu::tRDRAND) && X86Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = X86Features.has(Xbyak::util::Cpu::t3DN) && X86Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = X86Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = X86Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = X86Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = X86Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
SupportsCLWB = X86Features.has(Xbyak::util::Cpu::tCLWB);
|
||||
SupportsPMULL_128Bit = X86Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
SupportsAES256 = SupportsAES && X86Features.has(Xbyak::util::Cpu::tVAES);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0000, data);
|
||||
if (data[0] >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0008, data);
|
||||
SupportsCLZERO = data[1] & 1;
|
||||
}
|
||||
|
||||
SupportsAFP = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#endif
|
||||
#endif
|
||||
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
|
||||
|
||||
if (!Is64BitMode()) {
|
||||
///< Always disable AVX and AVX2 in 32-bit mode.
|
||||
// When AVX256 is enabled, signal frames start using significantly more stack space.
|
||||
// - 16bytes * 16 registers = 256 bytes for XMM registers.
|
||||
// - 32bytes * 16 registers = 512 bytes for YMM registers.
|
||||
// There are known game failures on real x86 hardware where a 32-bit game is running up against the wall on stack space on non-AVX
|
||||
// hardware and then explodes when run on AVX hardware. This is to guard against that.
|
||||
SupportsAVX = false;
|
||||
}
|
||||
|
||||
OverrideFeatures(this, ForceSVEWidth());
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -6,54 +6,59 @@
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static void LoadDeferredFCW(uint16_t NewFCW) {
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW) {
|
||||
softfloat_state State {};
|
||||
State.detectTininess = softfloat_tininess_afterRounding;
|
||||
State.exceptionFlags = 0;
|
||||
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
auto RC = (FCW >> 10) & 3;
|
||||
switch (RC) {
|
||||
case 0: softfloat_roundingMode = softfloat_round_near_even; break;
|
||||
case 1: softfloat_roundingMode = softfloat_round_min; break;
|
||||
case 2: softfloat_roundingMode = softfloat_round_max; break;
|
||||
case 3: softfloat_roundingMode = softfloat_round_minMag; break;
|
||||
case 0: State.roundingMode = softfloat_round_near_even; break;
|
||||
case 1: State.roundingMode = softfloat_round_min; break;
|
||||
case 2: State.roundingMode = softfloat_round_max; break;
|
||||
case 3: State.roundingMode = softfloat_round_minMag; break;
|
||||
}
|
||||
|
||||
return State;
|
||||
}
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, float src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, float src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t FCW, double src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
template<uint32_t Flags>
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
|
||||
bool eq, lt, nan;
|
||||
uint64_t ResultFlags = 0;
|
||||
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Flags & (1 << IR::FCMP_FLAG_LT) && lt) {
|
||||
X80SoftFloat::FCMP(&State, Src1, Src2, &eq, <, &nan);
|
||||
if (lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) && nan) {
|
||||
if (nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_EQ) && eq) {
|
||||
if (eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
return ResultFlags;
|
||||
@@ -62,292 +67,281 @@ struct OpHandlers<IR::OP_F80CMP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF64(&State);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI16(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI64(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
auto rv = extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX) {
|
||||
return INT16_MAX;
|
||||
} else if (rv < INT16_MIN) {
|
||||
if (rv > INT16_MAX || rv < INT16_MIN) {
|
||||
///< Indefinite value for 16-bit conversions.
|
||||
return INT16_MIN;
|
||||
} else {
|
||||
return rv;
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return extF80_to_i64(src, softfloat_round_minMag, false);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, X80SoftFloat src) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i64(&State, src, softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t FCW, int16_t src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, int32_t src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FRNDINT(Src1);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FRNDINT(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::F2XM1(Src1);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::F2XM1(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FTAN(Src1);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FTAN(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSQRT(Src1);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSQRT(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSIN(Src1);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSIN(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FCOS(Src1);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FCOS(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FADD(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FADD(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSUB(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSUB(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FMUL(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FMUL(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FDIV(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FDIV(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FYL2X(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FYL2X(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FATAN(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FATAN(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FREM1(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FREM1(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FREM(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FREM(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSCALE(Src1, Src2);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSCALE(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
double trunc = (double)(int64_t)(src2); // truncate
|
||||
return src1 * exp2(trunc);
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
}
|
||||
double trun = trunc(src2);
|
||||
return src1 * exp2(trun);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(Src1);
|
||||
Src1 = X80SoftFloat::FRNDINT(&State, Src1);
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
uint64_t Tmp = Src1.ToI64(&State);
|
||||
X80SoftFloat Rv;
|
||||
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
@@ -379,8 +373,7 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src) {
|
||||
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
|
||||
uint64_t BCD {};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
|
||||
@@ -35,14 +35,7 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC2] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t);
|
||||
Info[Core::OPINDEX_F80CMP_0] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>);
|
||||
Info[Core::OPINDEX_F80CMP_1] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>);
|
||||
Info[Core::OPINDEX_F80CMP_2] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>);
|
||||
Info[Core::OPINDEX_F80CMP_3] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>);
|
||||
Info[Core::OPINDEX_F80CMP_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>);
|
||||
Info[Core::OPINDEX_F80CMP_5] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>);
|
||||
Info[Core::OPINDEX_F80CMP_6] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>);
|
||||
Info[Core::OPINDEX_F80CMP_7] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>);
|
||||
Info[Core::OPINDEX_F80CMP] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle);
|
||||
Info[Core::OPINDEX_F80CVTTOINT_2] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2);
|
||||
Info[Core::OPINDEX_F80CVTTOINT_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4);
|
||||
|
||||
@@ -154,17 +147,8 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers {
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags),
|
||||
SupportsPreserveAllABI};
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
(Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP), SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -41,21 +42,6 @@ DEF_BINOP_WITH_CONSTANT(Lshl, lslv, lsl)
|
||||
DEF_BINOP_WITH_CONSTANT(Lshr, lsrv, lsr)
|
||||
DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
|
||||
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: {
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Src = GetRegPair(Op->Pair.ID());
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, Src.first);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.second, Src.second);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Constant) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
auto Dst = GetReg(Node);
|
||||
@@ -130,6 +116,21 @@ DEF_OP(AdcWithFlags) {
|
||||
adcs(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AdcZeroWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZeroWithFlags>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cset(Size, TMP1, ARMEmitter::Condition::CC_CC);
|
||||
adds(Size, GetReg(Node), GetReg(Op->Src1.ID()), TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(AdcZero) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZero>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cinc(Size, GetReg(Node), GetReg(Op->Src1.ID()), ARMEmitter::Condition::CC_CC);
|
||||
}
|
||||
|
||||
DEF_OP(Adc) {
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
|
||||
@@ -190,6 +191,30 @@ DEF_OP(TestNZ) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TestZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestZ>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < 4, "TestNZ used at higher sizes");
|
||||
const auto EmitSize = ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
uint64_t Mask = IROp->Size == 8 ? ~0ULL : ((1ull << (IROp->Size * 8)) - 1);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
|
||||
LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
if (Src1 == Src2) {
|
||||
tst(EmitSize, Src1 /* Src2 */, Mask);
|
||||
} else {
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
tst(EmitSize, TMP1, Mask);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
@@ -232,10 +257,8 @@ DEF_OP(CmpPairZ) {
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Compare, setting Z and clobbering NzCV
|
||||
const auto Src1 = GetRegPair(Op->Src1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Src2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, GetReg(Op->Src1Lo.ID()), GetReg(Op->Src2Lo.ID()));
|
||||
ccmp(EmitSize, GetReg(Op->Src1Hi.ID()), GetReg(Op->Src2Hi.ID()), ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
// Restore NzCV
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
@@ -274,8 +297,43 @@ DEF_OP(SetSmallNZV) {
|
||||
}
|
||||
|
||||
DEF_OP(AXFlag) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
|
||||
axflag();
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
axflag();
|
||||
} else {
|
||||
// AXFLAG is defined in the Arm spec as
|
||||
//
|
||||
// gt: nzCv -> nzCv
|
||||
// lt: Nzcv -> nzcv <==> 1 + 0
|
||||
// eq: nZCv -> nZCv <==> 1 + (~0)
|
||||
// un: nzCV -> nZcv <==> 0 + 0
|
||||
//
|
||||
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
|
||||
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
|
||||
auto V_inv = GetReg(IROp->Args[0].ID());
|
||||
csetm(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Condition::CC_EQ);
|
||||
ccmn(ARMEmitter::Size::i64Bit, V_inv, TMP1, ARMEmitter::StatusFlags {0x2} /* nzCv */, ARMEmitter::Condition::CC_LE);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Parity) {
|
||||
auto Op = IROp->C<IR::IROp_Parity>();
|
||||
auto Raw = GetReg(Op->Raw.ID());
|
||||
auto Dest = GetReg(Node);
|
||||
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, Raw, Raw, ARMEmitter::ShiftType::LSR, 4);
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 2);
|
||||
|
||||
if (Op->Invert) {
|
||||
eon(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
|
||||
} else {
|
||||
eor(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
|
||||
}
|
||||
|
||||
// The above sequence leaves garbage in the upper bits.
|
||||
if (Op->Mask) {
|
||||
and_(ARMEmitter::Size::i32Bit, Dest, Dest, 1);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
@@ -625,13 +683,17 @@ DEF_OP(ShiftFlags) {
|
||||
|
||||
// Set the output outside the branch to avoid needing an extra leg of the
|
||||
// branch. We specifically do not hardcode the PF register anywhere (relying
|
||||
// on a tied SRA register instead) to avoid fighting with RA/RCLSE.
|
||||
// on a tied SRA register instead) to avoid fighting with RA.
|
||||
if (PFTemp != PFInput) {
|
||||
mov(ARMEmitter::Size::i64Bit, PFTemp, PFInput);
|
||||
}
|
||||
|
||||
// We need to mask the source before comparing it. We don't just skip flag
|
||||
// updates for Src2=0 but anything that masks to zero.
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == 8 ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, Src2, &Done);
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
if (OpSize >= 4) {
|
||||
@@ -642,19 +704,27 @@ DEF_OP(ShiftFlags) {
|
||||
mov(ARMEmitter::Size::i64Bit, PFTemp, Dst);
|
||||
}
|
||||
|
||||
auto CFWord = TMP1;
|
||||
unsigned CFBit = 0;
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
if (Op->Shift == IR::ShiftType::LSL) {
|
||||
if (OpSize >= 4) {
|
||||
neg(EmitSize, TMP1, Src2);
|
||||
neg(EmitSize, CFWord, Src2);
|
||||
lsrv(EmitSize, CFWord, Src1, CFWord);
|
||||
} else {
|
||||
mov(EmitSize, TMP1, OpSize * 8);
|
||||
sub(EmitSize, TMP1, TMP1, Src2);
|
||||
CFWord = Dst.X();
|
||||
CFBit = (OpSize * 8);
|
||||
}
|
||||
} else {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, Src2, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, CFWord, Src2, 1);
|
||||
lsrv(EmitSize, CFWord, Src1, CFWord);
|
||||
}
|
||||
|
||||
lsrv(EmitSize, TMP1, Src1, TMP1);
|
||||
if (Op->InvertCF) {
|
||||
mvn(ARMEmitter::Size::i64Bit, TMP1, CFWord);
|
||||
CFWord = TMP1;
|
||||
}
|
||||
|
||||
bool SetOF = Op->Shift != IR::ShiftType::ASR;
|
||||
if (SetOF) {
|
||||
@@ -664,14 +734,20 @@ DEF_OP(ShiftFlags) {
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
rmif(TMP1, 63, (1 << 1) /* C */);
|
||||
rmif(CFWord, (CFBit - 1) % 64, (1 << 1) /* C */);
|
||||
|
||||
if (SetOF) {
|
||||
rmif(TMP3, OpSize * 8 - 1, (1 << 0) /* V */);
|
||||
}
|
||||
} else {
|
||||
mrs(TMP2, ARMEmitter::SystemRegister::NZCV);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP2, TMP1, 29 /* C */, 1);
|
||||
|
||||
if (CFBit != 0) {
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP1, CFWord, CFBit);
|
||||
CFWord = TMP1;
|
||||
}
|
||||
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP2, CFWord, 29 /* C */, 1);
|
||||
|
||||
if (SetOF) {
|
||||
lsr(EmitSize, TMP3, TMP3, OpSize * 8 - 1);
|
||||
@@ -689,6 +765,50 @@ DEF_OP(ShiftFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RotateFlags) {
|
||||
auto Op = IROp->C<IR::IROp_RotateFlags>();
|
||||
const auto Result = GetReg(Op->Result.ID());
|
||||
const auto Shift = GetReg(Op->Shift.ID());
|
||||
const bool Left = Op->Left;
|
||||
const auto EmitSize = Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = Op->Size * 8;
|
||||
unsigned CFBit = Left ? 0 : BitSize - 1;
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
// For ROL, OF is the LSB and MSB XOR'd together.
|
||||
// OF is architecturally only defined for 1-bit rotate.
|
||||
eor(ARMEmitter::Size::i64Bit, TMP1, Result, Result, ARMEmitter::ShiftType::LSR, Left ? BitSize - 1 : 1);
|
||||
unsigned OFBit = Left ? 0 : BitSize - 2;
|
||||
|
||||
// Invert result so we get inverted carry.
|
||||
mvn(ARMEmitter::Size::i64Bit, TMP2, Result);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
rmif(TMP2, (CFBit - 1) % 64, 1 << 1 /* nzCv */);
|
||||
rmif(TMP1, OFBit, 1 << 0 /* nzcV */);
|
||||
} else {
|
||||
if (OFBit != 0) {
|
||||
lsr(EmitSize, TMP1, TMP1, OFBit);
|
||||
}
|
||||
if (CFBit != 0) {
|
||||
lsr(EmitSize, TMP2, TMP2, CFBit);
|
||||
}
|
||||
|
||||
mrs(TMP3, ARMEmitter::SystemRegister::NZCV);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP3, TMP1, 28 /* V */, 1);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP3, TMP2, 29 /* C */, 1);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -704,59 +824,74 @@ DEF_OP(PDep) {
|
||||
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
// PDep implementation follows the ideas from
|
||||
// http://0x80.pl/articles/pdep-soft-emu.html ... Basically, iterate the *set*
|
||||
// bits only, which will be faster than the naive implementation as long as
|
||||
// there are enough holes in the mask.
|
||||
//
|
||||
// The specific arm64 assembly used is based on the sequence that clang
|
||||
// generates for the C code, giving context to the scheduling yielding better
|
||||
// ILP than I would do by hand. The registers are allocated by hand however,
|
||||
// to fit within the tight constraints we have here withot spilling. Also, we
|
||||
// use cbz/cbnz for conditional branching to avoid clobbering NZCV.
|
||||
|
||||
// We can't clobber these
|
||||
const auto OrigInput = GetReg(Op->Input.ID());
|
||||
const auto OrigMask = GetReg(Op->Mask.ID());
|
||||
|
||||
// So we have shadow as temporaries
|
||||
const auto Input = TMP1.R();
|
||||
const auto Mask = TMP2.R();
|
||||
if (CTX->HostFeatures.SupportsSVEBitPerm) {
|
||||
// SVE added support for PDEP but it needs to be done in a vector register.
|
||||
if (EmitSize == ARMEmitter::Size::i32Bit) {
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), OrigInput.W());
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP2.S(), OrigMask.W());
|
||||
bdep(ARMEmitter::SubRegSize::i32Bit, VTMP1.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(Dest, VTMP1, 0);
|
||||
} else {
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), OrigInput.X());
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), OrigMask.X());
|
||||
bdep(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(Dest, VTMP1, 0);
|
||||
}
|
||||
} else {
|
||||
// PDep implementation follows the ideas from
|
||||
// http://0x80.pl/articles/pdep-soft-emu.html ... Basically, iterate the *set*
|
||||
// bits only, which will be faster than the naive implementation as long as
|
||||
// there are enough holes in the mask.
|
||||
//
|
||||
// The specific arm64 assembly used is based on the sequence that clang
|
||||
// generates for the C code, giving context to the scheduling yielding better
|
||||
// ILP than I would do by hand. The registers are allocated by hand however,
|
||||
// to fit within the tight constraints we have here withot spilling. Also, we
|
||||
// use cbz/cbnz for conditional branching to avoid clobbering NZCV.
|
||||
|
||||
// these get used variously as scratch
|
||||
const auto T0 = TMP3.R();
|
||||
const auto T1 = TMP4.R();
|
||||
// So we have shadow as temporaries
|
||||
const auto Input = TMP1.R();
|
||||
const auto Mask = TMP2.R();
|
||||
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
// these get used variously as scratch
|
||||
const auto T0 = TMP3.R();
|
||||
const auto T1 = TMP4.R();
|
||||
|
||||
// First, copy the input/mask, since we'll be clobbering. Copy as 64-bit to
|
||||
// make this 0-uop on Firestorm.
|
||||
mov(ARMEmitter::Size::i64Bit, Input, OrigInput);
|
||||
mov(ARMEmitter::Size::i64Bit, Mask, OrigMask);
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
cbz(EmitSize, OrigMask, &Done);
|
||||
// First, copy the input/mask, since we'll be clobbering. Copy as 64-bit to
|
||||
// make this 0-uop on Firestorm.
|
||||
mov(ARMEmitter::Size::i64Bit, Input, OrigInput);
|
||||
mov(ARMEmitter::Size::i64Bit, Mask, OrigMask);
|
||||
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
and_(EmitSize, T0, T0, Mask);
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
cbz(EmitSize, OrigMask, &Done);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
sbfx(EmitSize, T1, Input, 0, 1);
|
||||
eor(EmitSize, Mask, Mask, T0);
|
||||
and_(EmitSize, T0, T1, T0);
|
||||
neg(EmitSize, T1, Mask);
|
||||
orr(EmitSize, Dest, Dest, T0);
|
||||
lsr(EmitSize, Input, Input, 1);
|
||||
and_(EmitSize, T0, Mask, T1);
|
||||
cbnz(EmitSize, T0, &NextBit);
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
and_(EmitSize, T0, T0, Mask);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
sbfx(EmitSize, T1, Input, 0, 1);
|
||||
eor(EmitSize, Mask, Mask, T0);
|
||||
and_(EmitSize, T0, T1, T0);
|
||||
neg(EmitSize, T1, Mask);
|
||||
orr(EmitSize, Dest, Dest, T0);
|
||||
lsr(EmitSize, Input, Input, 1);
|
||||
and_(EmitSize, T0, Mask, T1);
|
||||
cbnz(EmitSize, T0, &NextBit);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PExt) {
|
||||
@@ -769,35 +904,50 @@ DEF_OP(PExt) {
|
||||
const auto Mask = GetReg(Op->Mask.ID());
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
const auto MaskReg = TMP1;
|
||||
const auto BitReg = TMP2;
|
||||
const auto ValueReg = TMP3;
|
||||
if (CTX->HostFeatures.SupportsSVEBitPerm) {
|
||||
// SVE added support for PEXT but it needs to be done in a vector register.
|
||||
if (EmitSize == ARMEmitter::Size::i32Bit) {
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Input.W());
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP2.S(), Mask.W());
|
||||
bext(ARMEmitter::SubRegSize::i32Bit, VTMP1.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(Dest, VTMP1, 0);
|
||||
} else {
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Input.X());
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), Mask.X());
|
||||
bext(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(Dest, VTMP1, 0);
|
||||
}
|
||||
} else {
|
||||
const auto MaskReg = TMP1;
|
||||
const auto BitReg = TMP2;
|
||||
const auto ValueReg = TMP3;
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel EarlyExit;
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::SingleUseForwardLabel EarlyExit;
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
mov(EmitSize, ValueReg, Input);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
mov(EmitSize, ValueReg, Input);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
cbz(EmitSize, MaskReg, &Done);
|
||||
clz(EmitSize, BitReg, MaskReg);
|
||||
lslv(EmitSize, ValueReg, ValueReg, BitReg);
|
||||
lslv(EmitSize, MaskReg, MaskReg, BitReg);
|
||||
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
|
||||
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
|
||||
b(&NextBit);
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
cbz(EmitSize, MaskReg, &Done);
|
||||
clz(EmitSize, BitReg, MaskReg);
|
||||
lslv(EmitSize, ValueReg, ValueReg, BitReg);
|
||||
lslv(EmitSize, MaskReg, MaskReg, BitReg);
|
||||
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
|
||||
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
|
||||
b(&NextBit);
|
||||
|
||||
// Early exit
|
||||
Bind(&EarlyExit);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
// Early exit
|
||||
Bind(&EarlyExit);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
@@ -1325,11 +1475,6 @@ DEF_OP(Select) {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetRegPair(Op->Cmp1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Cmp2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, cc);
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
@@ -1390,6 +1535,12 @@ DEF_OP(NZCVSelect) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectIncrement) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectIncrement>();
|
||||
|
||||
csinc(ConvertSize(IROp), GetReg(Node), GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), MapCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1400,6 +1551,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1420,7 +1572,6 @@ DEF_OP(VExtractToGPR) {
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host doesn't support SVE. Cannot perform 256-bit operation.");
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
|
||||
|
||||
@@ -7,7 +7,8 @@ $end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
@@ -15,19 +15,43 @@ DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Expected = GetRegPair(Op->Expected.ID());
|
||||
auto Desired = GetRegPair(Op->Desired.ID());
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
auto Expected0 = GetReg(Op->ExpectedLo.ID());
|
||||
auto Expected1 = GetReg(Op->ExpectedHi.ID());
|
||||
auto Desired0 = GetReg(Op->DesiredLo.ID());
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP3, Expected.first);
|
||||
mov(EmitSize, TMP4, Expected.second);
|
||||
// RA has heuristics to try to pair sources, but we need to handle the cases
|
||||
// where they fail. We do so by moving to temporaries. Note we use 64-bit
|
||||
// moves here even for 32-bit cmpxchg, for the Firestorm register renamer.
|
||||
if (Desired1.Idx() != (Desired0.Idx() + 1) || Desired0.Idx() & 1) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, Desired0);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, Desired1);
|
||||
Desired0 = TMP1;
|
||||
Desired1 = TMP2;
|
||||
}
|
||||
|
||||
caspal(EmitSize, TMP3, TMP4, Desired.first, Desired.second, MemSrc);
|
||||
mov(EmitSize, Dst.first, TMP3.R());
|
||||
mov(EmitSize, Dst.second, TMP4.R());
|
||||
auto CaspalDst0 = Dst0;
|
||||
auto CaspalDst1 = Dst1;
|
||||
if (CaspalDst1.Idx() != (CaspalDst0.Idx() + 1) || CaspalDst0.Idx() & 1) {
|
||||
CaspalDst0 = TMP3;
|
||||
CaspalDst1 = TMP4;
|
||||
}
|
||||
|
||||
// We can't clobber the source, these moves are inherently required due to
|
||||
// ISA limitations. But by making them 64-bit, Firestorm can rename.
|
||||
mov(ARMEmitter::Size::i64Bit, CaspalDst0, Expected0);
|
||||
mov(ARMEmitter::Size::i64Bit, CaspalDst1, Expected1);
|
||||
caspal(EmitSize, CaspalDst0, CaspalDst1, Desired0, Desired1, MemSrc);
|
||||
|
||||
if (CaspalDst0 != Dst0) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst0, CaspalDst0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst1, CaspalDst1);
|
||||
}
|
||||
} else {
|
||||
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
|
||||
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
|
||||
@@ -43,19 +67,19 @@ DEF_OP(CASPair) {
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected.first);
|
||||
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, TMP2, Expected0);
|
||||
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxp(EmitSize, TMP2, Desired.first, Desired.second, MemSrc);
|
||||
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
mov(EmitSize, Dst.first, Expected.first);
|
||||
mov(EmitSize, Dst.second, Expected.second);
|
||||
mov(EmitSize, Dst0, Expected0);
|
||||
mov(EmitSize, Dst1, Expected1);
|
||||
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst.first, TMP2.R());
|
||||
mov(EmitSize, Dst.second, TMP3.R());
|
||||
mov(EmitSize, Dst0, TMP2.R());
|
||||
mov(EmitSize, Dst1, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
|
||||
@@ -11,11 +11,11 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -78,8 +78,12 @@ DEF_OP(ExitFunction) {
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
@@ -112,20 +116,27 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ || Op->Cond.Val == FEXCore::IR::COND_NEQ, "CondJump: Expected simple "
|
||||
"condition");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
|
||||
}
|
||||
|
||||
// TODO: Wire up tbz/tbnz
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
@@ -184,7 +195,7 @@ DEF_OP(Syscall) {
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask);
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
@@ -255,15 +266,13 @@ DEF_OP(InlineSyscall) {
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg == ARMEmitter::Reg::r8) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
} else if (Reg == ARMEmitter::Reg::r4) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
|
||||
} else if (Reg == ARMEmitter::Reg::r5) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RCX]));
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
@@ -285,7 +294,7 @@ DEF_OP(InlineSyscall) {
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r1);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
@@ -427,10 +436,11 @@ DEF_OP(CPUID) {
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
|
||||
// Results want to be 4xi32 scalars
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX.ID()), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX.ID()), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP2, 32, 32);
|
||||
}
|
||||
|
||||
DEF_OP(XGetBV) {
|
||||
@@ -459,11 +469,9 @@ DEF_OP(XGetBV) {
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP1, 32, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -16,6 +16,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
@@ -111,6 +112,8 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -203,6 +206,7 @@ DEF_OP(Vector_SToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -235,6 +239,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -265,6 +270,8 @@ DEF_OP(Vector_FToS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -294,6 +301,8 @@ DEF_OP(Vector_FToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -395,6 +404,7 @@ DEF_OP(Vector_FToI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -449,5 +459,64 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_F64ToI32) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_F64ToI32>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto Mask = Is256Bit ? PRED_TMP_32B.Merging() : PRED_TMP_16B.Merging();
|
||||
// First step is to round the f64 values to integrals (frint*)
|
||||
// Then convert to integers using fcvtzs.
|
||||
auto CVTReg = Dst.Z();
|
||||
switch (Round) {
|
||||
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
|
||||
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
|
||||
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
///< Fixup format of register that fcvtzs returns.
|
||||
uzp1(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
if (Op->EnsureZeroUpperHalf) {
|
||||
///< Match CVTPD2DQ/CVTTPD2DQ behaviour if necessary by zeroing the upper bits here.
|
||||
if (Is256Bit) {
|
||||
mov(Dst.Q(), Dst.Q());
|
||||
} else {
|
||||
mov(Dst.D(), Dst.D());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// This has a known precision issue that isn't easily resolvable without throwing away performance.
|
||||
// Doing the conversion in multi-stage steps has an issue that you can lose precision in the f32->i32 step if your source was f64.
|
||||
// To get around this with ASIMD FEX needs to use fcvtzs (Scalar, Integer, to GPR) for each F64 to be directly converted to i32.
|
||||
// This is a very costly transform that the SVE path doesn't need to do since it supports f64->i32 directly.
|
||||
// If this precision issue is necessary then we can add an option for it in the future.
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
switch (Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
// Now narrow from f64 to f32.
|
||||
fcvtn(ARMEmitter::SubRegSize::i32Bit, Dst.Q(), Dst.Q());
|
||||
|
||||
///< Convert the two F32 integrals to real integers.
|
||||
fcvtzs(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,18 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Value.ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -47,8 +47,8 @@ static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LDIV(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
static int64_t LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
}
|
||||
@@ -59,8 +59,8 @@ static uint64_t LUREM(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
static int64_t LREM(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
@@ -496,7 +496,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
|
||||
auto offset = HostCode / 4 - branch / 4;
|
||||
if (vixl::IsInt26(offset)) {
|
||||
if (ARMEmitter::Emitter::IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
ARMEmitter::Emitter emit((uint8_t*)(branch), 4);
|
||||
@@ -654,12 +654,6 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
@@ -724,12 +718,22 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (vixl::aarch64::Assembler::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
|
||||
@@ -872,7 +876,7 @@ void Arm64JITCore::ResetStack() {
|
||||
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (vixl::aarch64::Assembler::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
@@ -885,11 +889,4 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl*
|
||||
return fextl::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsFlags = true,
|
||||
.SupportsVTBL2 = true,
|
||||
};
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -14,9 +14,6 @@ $end_info$
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -40,25 +37,10 @@ public:
|
||||
explicit Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::string GetName() override {
|
||||
return "JIT";
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
|
||||
[[nodiscard]]
|
||||
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool NeedsOpDispatch() override {
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
void ClearRelocations() override {
|
||||
@@ -67,8 +49,6 @@ public:
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
@@ -114,15 +94,6 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return std::make_pair(GeneralRegisters[Reg.Reg], GeneralRegisters[Reg.Reg + 1]);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
@@ -253,8 +224,6 @@ private:
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
@@ -345,6 +314,11 @@ private:
|
||||
uint32_t SpillSlots {};
|
||||
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node);
|
||||
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
void VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Upper, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2,
|
||||
ARMEmitter::VRegister Addend);
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
@@ -352,6 +326,10 @@ private:
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
void Emulate128BitGather(size_t Size, size_t ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, size_t VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
|
||||
Loaded 100 of 492 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user