mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 20:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b3bc1e23cc | ||
|
|
1a9b6a89f4 | ||
|
|
21bf35d211 | ||
|
|
9473025b18 | ||
|
|
02f15f4099 | ||
|
|
e007789ced | ||
|
|
a2b165043c | ||
|
|
5b5808218b | ||
|
|
41ec987f3e | ||
|
|
0f4a5edf4f | ||
|
|
03f73531d3 | ||
|
|
69181d438c | ||
|
|
a2cbfccb3b | ||
|
|
4e01452a65 | ||
|
|
cc7a56b1a6 | ||
|
|
0b0dd3891e | ||
|
|
8bc33e95c1 | ||
|
|
96a0364a86 | ||
|
|
c0a783997d | ||
|
|
f78537109d | ||
|
|
0c156ed6f9 | ||
|
|
920913cf80 | ||
|
|
8840b2154c | ||
|
|
e02be8073e | ||
|
|
f75d3550b4 | ||
|
|
802c588695 | ||
|
|
fd962f40d7 | ||
|
|
a9b660af69 | ||
|
|
fd5c36ba9c | ||
|
|
5be798e9e6 | ||
|
|
09997cff9c | ||
|
|
c9d1f0d75a | ||
|
|
95b7592241 | ||
|
|
1dc4f8c429 | ||
|
|
1d7fcdb54a | ||
|
|
45d3b83143 | ||
|
|
d97fa9af14 | ||
|
|
c9101d3f68 | ||
|
|
52f64a0c7b | ||
|
|
737f917838 | ||
|
|
a6c6248bcb | ||
|
|
5646428640 | ||
|
|
0c8df2beaf | ||
|
|
de0f3984e9 | ||
|
|
ada226bbb4 | ||
|
|
4bc5a09e62 | ||
|
|
5704b5f23f | ||
|
|
6017a9135a | ||
|
|
6ef6d9c391 | ||
|
|
0ad6f98a8c | ||
|
|
1354f92cc5 | ||
|
|
8b90caad95 | ||
|
|
3a4a965347 | ||
|
|
45cdab2ac3 | ||
|
|
b89dc56ae1 | ||
|
|
d675b4af6f | ||
|
|
d75fb38344 | ||
|
|
4cb385a27b | ||
|
|
9a4fdd8059 | ||
|
|
363411f0c7 | ||
|
|
5bc418407c | ||
|
|
46e2dc7498 | ||
|
|
4a54197868 | ||
|
|
3f214dd244 | ||
|
|
61ca651fe1 | ||
|
|
cd0a340d29 | ||
|
|
77e8be1215 | ||
|
|
182010ca97 | ||
|
|
f7c663240e | ||
|
|
e9244680aa | ||
|
|
82b4aef30d | ||
|
|
22919a5b65 | ||
|
|
00dc373bb9 | ||
|
|
e593237670 | ||
|
|
0a4bf10ba5 | ||
|
|
f47caf48c6 | ||
|
|
5674d3a871 | ||
|
|
fde64aedf7 | ||
|
|
e03b859c20 | ||
|
|
ce4e991a6e | ||
|
|
7d822ba1c8 | ||
|
|
f90dcd2eb1 | ||
|
|
adbdd33ece | ||
|
|
06250d806d | ||
|
|
613ed559e7 | ||
|
|
8ac3841946 | ||
|
|
458259bf47 | ||
|
|
dc65a5ef8c | ||
|
|
2fc529d5b7 | ||
|
|
ea489567da | ||
|
|
ed69eb9f6f | ||
|
|
6eae064511 | ||
|
|
f1eb98548a | ||
|
|
86af1f6a68 | ||
|
|
2d4bf97cac | ||
|
|
542aeed7b9 | ||
|
|
f7d827a26a | ||
|
|
37b5bc49c6 | ||
|
|
dcb3f182d6 | ||
|
|
ba45bf4ae7 | ||
|
|
121d9fda2d | ||
|
|
73ede9d000 | ||
|
|
6dfea8a80f | ||
|
|
f06972c9e2 | ||
|
|
ef6c220a75 | ||
|
|
1e4a6d432c | ||
|
|
4ebd180147 | ||
|
|
ac4ef63ae6 | ||
|
|
65fa495890 | ||
|
|
b2392ef1c6 | ||
|
|
8e4d52396b | ||
|
|
6eeb45b2dc | ||
|
|
22cf2696da | ||
|
|
468f7471e1 | ||
|
|
8081ac61e5 | ||
|
|
435b4daae1 | ||
|
|
30974fb2c9 | ||
|
|
5ee913bc75 | ||
|
|
db706bb28f | ||
|
|
0d1810c159 | ||
|
|
c1dab3e6cc | ||
|
|
e8e70c4faf | ||
|
|
88247141d7 | ||
|
|
f502154f96 | ||
|
|
7a59fb3e25 | ||
|
|
e71f3e898f | ||
|
|
8369f9c25b | ||
|
|
456e9dbdea | ||
|
|
41d00c8dc6 | ||
|
|
886c562882 | ||
|
|
8be2ad8c69 | ||
|
|
b2a3c6a043 | ||
|
|
e2db607769 | ||
|
|
590422b295 | ||
|
|
7552ad29fa | ||
|
|
9d268df91f | ||
|
|
699541485d | ||
|
|
46a63186a2 | ||
|
|
520441c262 | ||
|
|
90f347839d | ||
|
|
c9e7d9f331 | ||
|
|
8c3a3bfb7c | ||
|
|
9034946b43 | ||
|
|
9432a84cb4 | ||
|
|
056f44be0b | ||
|
|
b5420f5db3 | ||
|
|
c94268789b | ||
|
|
af15277fc4 | ||
|
|
86e09a00f0 | ||
|
|
238ffdf893 | ||
|
|
c94721a04b | ||
|
|
c87f361bb5 | ||
|
|
ea52ae3bbc | ||
|
|
0fa4390e47 | ||
|
|
7a774a8d80 | ||
|
|
361e684c64 | ||
|
|
4c74913edf | ||
|
|
059472fcef | ||
|
|
4a11111abd | ||
|
|
f673afc38f | ||
|
|
1fad26d72f | ||
|
|
2b5ddb6b93 | ||
|
|
c140dd7da8 | ||
|
|
da126141d3 | ||
|
|
874ae5b0fc | ||
|
|
35f192b6fd | ||
|
|
651c6f8ddf | ||
|
|
73ca4e5687 | ||
|
|
cb9cc74fcc | ||
|
|
512d6d0069 | ||
|
|
d1116456fc | ||
|
|
84985952c9 | ||
|
|
a351620c60 | ||
|
|
9117f7e724 | ||
|
|
6e52a16ef3 | ||
|
|
8e391e7a61 | ||
|
|
98fbc4a46d | ||
|
|
8481aeccb5 | ||
|
|
b1df63f425 | ||
|
|
a12802e74c | ||
|
|
39c73d975b | ||
|
|
30cb1aaaed | ||
|
|
cbf41448fc | ||
|
|
47bdc9af12 | ||
|
|
0fad5b88c1 | ||
|
|
fb93fa573c | ||
|
|
8c9fe0dd31 | ||
|
|
dda3afcfaf | ||
|
|
34ceefb2c3 | ||
|
|
25ef63a069 | ||
|
|
99a9c88f3f | ||
|
|
77f56199e8 | ||
|
|
6c13b629af | ||
|
|
3ebe9f7b04 | ||
|
|
1979273ce5 | ||
|
|
005389f8c1 | ||
|
|
a33443db62 | ||
|
|
68599bf124 | ||
|
|
d7f9c7ece2 | ||
|
|
4bffdc6345 | ||
|
|
892c07a5ed | ||
|
|
780491d61b | ||
|
|
cbe55b0765 | ||
|
|
2ff5096103 | ||
|
|
797737a84d | ||
|
|
cfc1aa593b | ||
|
|
fbc5d583a2 | ||
|
|
6b964f70e0 | ||
|
|
78844ee975 | ||
|
|
51afcb7143 | ||
|
|
5258b1972b | ||
|
|
c6616d64d8 | ||
|
|
d9b9ce804b | ||
|
|
132aa7e4d3 | ||
|
|
0419d065b5 | ||
|
|
fc00a31aee | ||
|
|
1de84110e8 | ||
|
|
1962f036e1 | ||
|
|
879a081556 | ||
|
|
105060363f | ||
|
|
bbb3a6439f | ||
|
|
40b67462b7 | ||
|
|
d853de39ff | ||
|
|
401d89ee40 | ||
|
|
75a62f856b | ||
|
|
e8aaadb2d0 | ||
|
|
27c03f98d8 | ||
|
|
bfd606ec3d | ||
|
|
9823a64164 | ||
|
|
72483ea21d | ||
|
|
1a91d849f0 | ||
|
|
6d4cef723a | ||
|
|
1c2fd72c84 | ||
|
|
77f2378080 | ||
|
|
1cc9f2107d | ||
|
|
b979b339fc | ||
|
|
46e5343a0e | ||
|
|
d519883dbe | ||
|
|
beb2c36fc2 | ||
|
|
f45ea1e0f6 | ||
|
|
92593162b0 | ||
|
|
307158d425 | ||
|
|
5c62ea21f4 | ||
|
|
73caa6725f | ||
|
|
b0b23abedd | ||
|
|
f55a653e99 | ||
|
|
4cb35d2f5e | ||
|
|
7e4334c98a | ||
|
|
0d0b99f344 | ||
|
|
44e06185b7 | ||
|
|
3f89cf1512 | ||
|
|
18154183ad | ||
|
|
49abe8afb5 | ||
|
|
7916281ee7 | ||
|
|
d7bc0370ee | ||
|
|
c5da0e7ac1 | ||
|
|
0e007d2724 | ||
|
|
63b31d54c4 | ||
|
|
466edf7744 | ||
|
|
0bb59ade53 | ||
|
|
c03ed529e6 | ||
|
|
f806ca688c | ||
|
|
48531e1dd2 | ||
|
|
1864a1d3b5 | ||
|
|
e98a46aa5f | ||
|
|
96e9b5cf51 | ||
|
|
265c918d90 | ||
|
|
76a5e4e66e | ||
|
|
dce6389d87 | ||
|
|
46b306e861 | ||
|
|
32d7fae373 | ||
|
|
47e5096676 | ||
|
|
f25b0461d0 | ||
|
|
11402b637a | ||
|
|
a9c27646a0 | ||
|
|
278f411cfa | ||
|
|
cafbcfec69 | ||
|
|
f5ed9c4ff3 | ||
|
|
e027521941 | ||
|
|
cb6680ef57 | ||
|
|
613a368f5f | ||
|
|
13500e59a3 | ||
|
|
1306e597dd | ||
|
|
87f8a6655a | ||
|
|
4d70f4fc4e | ||
|
|
9dd715573c | ||
|
|
87772efb31 | ||
|
|
bb922f9a9e | ||
|
|
7fe14b0748 | ||
|
|
4ab822aebb | ||
|
|
ecb7956d78 | ||
|
|
e232a10442 | ||
|
|
efada6a0ea | ||
|
|
cfbd74e17b | ||
|
|
4eb91ef7f1 | ||
|
|
2d18156e15 | ||
|
|
dff0b45f29 | ||
|
|
8fb7e8d80e | ||
|
|
60fe987e09 | ||
|
|
7180bb1496 | ||
|
|
001a086d85 | ||
|
|
8a711383bb | ||
|
|
257a3a54dc | ||
|
|
e4fadd6992 | ||
|
|
9e5971b89c | ||
|
|
28c168ea0c | ||
|
|
f944709139 | ||
|
|
e8bf7a1a46 | ||
|
|
f87b00a7e4 | ||
|
|
a9f0cb15bf | ||
|
|
a78860c194 | ||
|
|
f056cc790e | ||
|
|
aac4e25ca4 | ||
|
|
546a1edb55 | ||
|
|
21838fe03f | ||
|
|
42e2a5421a | ||
|
|
557df4f0a7 | ||
|
|
99ba648a71 | ||
|
|
97daec3dba | ||
|
|
4f20dba505 | ||
|
|
2990a9d820 | ||
|
|
53bbbd5a4f | ||
|
|
047dddb023 | ||
|
|
4f66ff6ec4 | ||
|
|
e46dbf7b7e | ||
|
|
067f807405 | ||
|
|
463b4b748c | ||
|
|
3eae668cec | ||
|
|
fc8bf9f0f6 | ||
|
|
3cfc1de410 | ||
|
|
a14353bc3c | ||
|
|
629e547e5f | ||
|
|
12b710ed16 | ||
|
|
ea275bbdcc | ||
|
|
e6a48ad7f3 | ||
|
|
114b626cff | ||
|
|
4d35d550c1 | ||
|
|
e8c1dfa03a | ||
|
|
9d04c4daee | ||
|
|
1ec31c610c | ||
|
|
86f8ebf0ee | ||
|
|
bbd0d26c16 | ||
|
|
1eac7e7105 | ||
|
|
1f9458a3c3 | ||
|
|
a3be4b77fa | ||
|
|
4cc14cf0e9 | ||
|
|
b2ec28503d | ||
|
|
170c9ee9e4 | ||
|
|
c77c2faaea | ||
|
|
1eb36b8b31 | ||
|
|
465ecd9b19 | ||
|
|
df354e37dd | ||
|
|
43e6d398b6 | ||
|
|
0d7c856775 | ||
|
|
141dddc83e | ||
|
|
64aa3bfabe | ||
|
|
f02a111d33 | ||
|
|
79f7baffe3 | ||
|
|
7150c532f3 | ||
|
|
e48fb1850e | ||
|
|
88dba60bee | ||
|
|
c9fb9c4cae | ||
|
|
d615ae9c6a | ||
|
|
7629edcf61 | ||
|
|
73d250c555 | ||
|
|
337f8b06a3 | ||
|
|
87fa545bd0 | ||
|
|
7747ac8de8 | ||
|
|
deb1c9e933 | ||
|
|
672a88395d | ||
|
|
88524ce718 | ||
|
|
bb153054f9 | ||
|
|
22a7a49042 | ||
|
|
9876f3eb5c | ||
|
|
b9e4ce4029 | ||
|
|
0c048772e0 | ||
|
|
cf66643c60 | ||
|
|
830c1884d1 | ||
|
|
5abf9de8a5 | ||
|
|
fa21944428 | ||
|
|
5cdde0bef1 | ||
|
|
25960fe6b1 | ||
|
|
eb8626c1f7 | ||
|
|
100b4d4a5b | ||
|
|
ef7853ca4a | ||
|
|
55d65f3aea | ||
|
|
2c2abc550b | ||
|
|
a1dc132f03 | ||
|
|
719803bc5a | ||
|
|
ef31e0c7c7 | ||
|
|
eecd016ba8 | ||
|
|
b2ec6d5208 | ||
|
|
416a7b825d | ||
|
|
c3511ffa48 | ||
|
|
b9c3277b09 | ||
|
|
09f259c458 | ||
|
|
7a2d23e189 | ||
|
|
1e10561892 | ||
|
|
dcc8d2a1c7 | ||
|
|
9183cf144f | ||
|
|
043a03547f | ||
|
|
7022b3b825 | ||
|
|
6016c48f98 | ||
|
|
60408241c5 | ||
|
|
bed10000e4 | ||
|
|
a899444ddd | ||
|
|
72df6a26d5 | ||
|
|
dfeee6a1ed | ||
|
|
91d6dc0528 | ||
|
|
54e784742c | ||
|
|
22936144ce | ||
|
|
fea0162096 | ||
|
|
8287375117 | ||
|
|
0106f03b44 | ||
|
|
3e2d1cf8c0 | ||
|
|
392438d70a | ||
|
|
0930a10026 | ||
|
|
cb8ab3d328 | ||
|
|
a670c1d762 | ||
|
|
91f056bd2d | ||
|
|
efba2aed7d | ||
|
|
233aef5289 | ||
|
|
9d38bc0c81 | ||
|
|
a76967650a | ||
|
|
7f0cccfbe0 | ||
|
|
437ea926b4 | ||
|
|
4359f7439a | ||
|
|
a1712ab455 | ||
|
|
0553f69eed | ||
|
|
d03dfe2b33 | ||
|
|
5e105e86af | ||
|
|
97bc3afb84 | ||
|
|
bccc9f1656 | ||
|
|
834e862dfb | ||
|
|
b433d7b4cb | ||
|
|
b6d36f123a | ||
|
|
0ab2a550b1 | ||
|
|
1877457c4d | ||
|
|
e5867b89ed | ||
|
|
487785de41 | ||
|
|
e741ac37b1 | ||
|
|
5d928fbdee | ||
|
|
d1a14bf321 | ||
|
|
3f98eff6e5 | ||
|
|
46d92b7d78 | ||
|
|
4f6e12c460 | ||
|
|
45493ad3ef | ||
|
|
fe886716a4 | ||
|
|
e9a0d95c65 | ||
|
|
bb0757c2c9 | ||
|
|
24101d99ac | ||
|
|
7d226a6b18 | ||
|
|
5143579c20 | ||
|
|
a649e6aedf | ||
|
|
3c021d64f8 | ||
|
|
5ed0028d1a | ||
|
|
61854c3655 | ||
|
|
f77f243ae6 | ||
|
|
5092675179 | ||
|
|
9883f8fced | ||
|
|
e3f6ef6b18 | ||
|
|
606242472a | ||
|
|
8941b8a312 | ||
|
|
37832af818 | ||
|
|
a7334608fa | ||
|
|
bc42e849e9 | ||
|
|
54c33b07a2 | ||
|
|
a5b034129e | ||
|
|
8d57446e88 | ||
|
|
bba8716c7e | ||
|
|
1771d086ed | ||
|
|
d416331650 | ||
|
|
aea90287eb | ||
|
|
95a4994200 | ||
|
|
e30f3bd0c8 | ||
|
|
bd3facde63 | ||
|
|
75e8b6cab8 | ||
|
|
91330d6b86 | ||
|
|
92f2dc0737 | ||
|
|
5b97e6d354 | ||
|
|
728d19dbbe | ||
|
|
4b9c7bd116 | ||
|
|
017f101448 | ||
|
|
4bde9f796a | ||
|
|
7b790903c9 | ||
|
|
9c62442f43 | ||
|
|
a9a2c95202 | ||
|
|
194f97aefa | ||
|
|
1c0a0c1fe5 | ||
|
|
149fd2a2b6 | ||
|
|
06211d330d | ||
|
|
18c9e84543 | ||
|
|
a4831519cf | ||
|
|
bac5eff296 | ||
|
|
97565b68db | ||
|
|
6b053d298e | ||
|
|
823222702a | ||
|
|
e8132b82d0 | ||
|
|
cdac296ce3 | ||
|
|
ffbcc52af6 | ||
|
|
ddafc172fc | ||
|
|
670a029228 | ||
|
|
30dcf2cf39 | ||
|
|
df17f7adbb | ||
|
|
9ffbffe463 | ||
|
|
1aaa03a10d | ||
|
|
e4b2ed72a8 | ||
|
|
c6d1b645bd | ||
|
|
35d126e12d | ||
|
|
d7a43c5553 | ||
|
|
8ccd95ff02 | ||
|
|
af91430007 | ||
|
|
eadce2854f | ||
|
|
29963ad5e2 | ||
|
|
335aedc781 | ||
|
|
41477db7aa | ||
|
|
02e245d61f | ||
|
|
03724c8486 | ||
|
|
f8a575a982 | ||
|
|
2d05ffe3aa | ||
|
|
042b511126 | ||
|
|
8b0f66d599 | ||
|
|
62b5441d56 | ||
|
|
8b20921341 | ||
|
|
8633528cee | ||
|
|
ce961a3ef4 | ||
|
|
8ce87d3b27 | ||
|
|
24f03dd740 | ||
|
|
a1b0d1853b | ||
|
|
6531d369cd | ||
|
|
7344680672 | ||
|
|
16ae9ad1c7 | ||
|
|
737400fd84 | ||
|
|
331cf66fda | ||
|
|
96c35a7bc7 | ||
|
|
15903f5300 | ||
|
|
a8ed2af658 | ||
|
|
3a81efdb28 | ||
|
|
3275dabd85 | ||
|
|
64c45ed70a | ||
|
|
02d7f68094 | ||
|
|
afdc037110 | ||
|
|
6e0a1a5f14 | ||
|
|
ad332e3c30 | ||
|
|
1aeab04c9f | ||
|
|
aba42570a1 | ||
|
|
7f243ee08a | ||
|
|
82f7bb282f | ||
|
|
1248f3573c | ||
|
|
d89f53cfd3 | ||
|
|
89f1e61779 | ||
|
|
11d396ca06 | ||
|
|
3ff7cf19c0 | ||
|
|
11bbfe5bed | ||
|
|
07aa5327f4 | ||
|
|
c6ba51ee35 | ||
|
|
cd40a85567 | ||
|
|
19125720c2 | ||
|
|
9b70c1d4ac | ||
|
|
02d06ee833 | ||
|
|
5e8d9601ed | ||
|
|
640cfcd0bd | ||
|
|
5d71dfed86 | ||
|
|
1c3af30e96 | ||
|
|
a06fc3641f | ||
|
|
5cbf984743 | ||
|
|
d1ece88bed | ||
|
|
b1a00b05c4 | ||
|
|
1087c45b6c | ||
|
|
d9da63d492 | ||
|
|
597ffe5ba6 | ||
|
|
6517f7eb30 | ||
|
|
73ff932bb6 | ||
|
|
9b742b1fac | ||
|
|
8c63afb345 | ||
|
|
c14f4355b7 | ||
|
|
6645e68c95 | ||
|
|
ff9de85851 | ||
|
|
4dca609614 | ||
|
|
4fc365c149 | ||
|
|
8804e0a62a | ||
|
|
1f512b098e | ||
|
|
a31c3ad87a | ||
|
|
e92cfc4d7a | ||
|
|
820d743321 | ||
|
|
3b74e34b47 | ||
|
|
054430f004 | ||
|
|
aeff4346e8 | ||
|
|
5b51440c87 | ||
|
|
baace30e30 | ||
|
|
11c6f97643 | ||
|
|
1f44037d9a | ||
|
|
3bc484b586 | ||
|
|
79170a23e5 | ||
|
|
81e52ca19f | ||
|
|
5301036968 | ||
|
|
2321d2ad97 | ||
|
|
9c2d29ee4e | ||
|
|
4b8ac0f8c6 | ||
|
|
70c93efdfa | ||
|
|
f5e207c28f | ||
|
|
9c64a22280 | ||
|
|
5f57fa4b77 | ||
|
|
fe81523179 | ||
|
|
4be61aaaaf | ||
|
|
da0681ca81 | ||
|
|
02258073e3 | ||
|
|
cde55d44ef | ||
|
|
97c12ef351 | ||
|
|
94e0591601 | ||
|
|
b802f64ffe | ||
|
|
0e32d8a8b6 | ||
|
|
fe8bd17c88 | ||
|
|
f333068d9a | ||
|
|
3bd4b23af7 | ||
|
|
060745f98b | ||
|
|
77ae7a8a3f | ||
|
|
b989bb9569 |
No files matched your search
@@ -14,6 +14,7 @@ env:
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -24,7 +25,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -240,13 +241,19 @@ jobs:
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
name: GLIBC fault test
|
||||
# This workflow file is the same as the `Build + Test` with some key differences
|
||||
# - Runs on any x86 and ARM64 runner
|
||||
# - Disables the glibc jemalloc compile option
|
||||
# - Enables the glibc allocator fault option
|
||||
# - Disables gvisor tests to reduce stress on CI machines (tmp/shm tests overwhelm them)
|
||||
# - Disables thunk tests since they are incompatible with glibc fault allocator
|
||||
# - Disables ARMEmitter tests (We don't want to fault test vixl's disassembler)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
# Run on an x86 device and any ARM runner.
|
||||
arch: [[self-hosted, x64], [self-hosted, ARM64]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Posix Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the posixtest
|
||||
run: cmake --build . --config $BUILD_TYPE --target posix_tests
|
||||
|
||||
- name: Posix Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
|
||||
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
|
||||
|
||||
- name: GCC64 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
|
||||
|
||||
- name: gcc target tests 32
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
|
||||
|
||||
- name: GCC32 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
|
||||
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -24,7 +25,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -110,7 +111,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
+5
-2
@@ -3,7 +3,7 @@
|
||||
path = External/vixl
|
||||
url = https://github.com/FEX-Emu/vixl.git
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = External/cpp-optparse
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/imgui"]
|
||||
path = External/imgui
|
||||
@@ -48,8 +48,11 @@
|
||||
[submodule "External/robin-map"]
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/Tessil/robin-map.git
|
||||
url = https://github.com/FEX-Emu/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
[submodule "External/jemalloc_glibc"]
|
||||
path = External/jemalloc_glibc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
+32
-4
@@ -18,10 +18,10 @@ option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
@@ -33,11 +33,19 @@ option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FA
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
message (STATUS "Mingw build")
|
||||
set (MINGW_BUILD TRUE)
|
||||
set (ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
@@ -49,6 +57,14 @@ if (ENABLE_FEXCORE_PROFILER)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC AND ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
message(FATAL_ERROR "Can't have both glibc fault allocator and jemalloc glibc allocator enabled at the same time")
|
||||
endif()
|
||||
|
||||
if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
add_definitions(-DGLIBC_ALLOCATOR_FAULT=1)
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
@@ -178,7 +194,22 @@ if (ENABLE_TSAN)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
# The glibc jemalloc subproject which hooks the glibc allocator.
|
||||
# Required for thunks to work.
|
||||
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
|
||||
add_definitions(-DENABLE_JEMALLOC_GLIBC=1)
|
||||
add_subdirectory(External/jemalloc_glibc/)
|
||||
else()
|
||||
message (STATUS
|
||||
" jemalloc glibc allocator disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
" This will very explicitly break thunk execution!\n"
|
||||
" Use at your own risk!")
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
# The jemalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
@@ -230,9 +261,6 @@ if (BUILD_TESTS)
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/cpp-optparse/)
|
||||
include_directories(External/cpp-optparse/)
|
||||
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
|
||||
@@ -173,6 +173,14 @@
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
Vendored
+1
@@ -77,6 +77,7 @@ configure_file(
|
||||
|
||||
include_directories(${CMAKE_BINARY_DIR}/generated)
|
||||
|
||||
add_compile_options(-fno-exceptions)
|
||||
add_subdirectory(Source/)
|
||||
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
|
||||
+10
-7
@@ -22,10 +22,10 @@ def print_header():
|
||||
#define OPT_UINT64(group, enum, json, default) OPT_BASE(uint64_t, group, enum, json, default)
|
||||
#endif
|
||||
#ifndef OPT_STR
|
||||
#define OPT_STR(group, enum, json, default) OPT_BASE(std::string, group, enum, json, default)
|
||||
#define OPT_STR(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#endif
|
||||
#ifndef OPT_STRARRAY
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(std::string, group, enum, json, default)
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#endif
|
||||
|
||||
'''
|
||||
@@ -371,13 +371,16 @@ def print_parse_argloader_options(options):
|
||||
|
||||
value_type = op_vals["Type"]
|
||||
NeedsString = False
|
||||
conversion_func = "std::to_string"
|
||||
conversion_func = "fextl::fmt::format(\"{}\", "
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = ""
|
||||
conversion_func = "("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
@@ -387,11 +390,11 @@ def print_parse_argloader_options(options):
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
output_argloader.write("\tstd::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
else:
|
||||
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
|
||||
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}(UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
+51
-20
@@ -1,13 +1,19 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Common/Paths.cpp
|
||||
Interface/Config/Config.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/CPUInfo.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list(APPEND FEXCORE_BASE_SRCS
|
||||
Utils/Allocator/64BitAllocator.cpp)
|
||||
endif()
|
||||
|
||||
set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Common/SoftFloat-3e/extF80_add.c
|
||||
@@ -99,12 +105,11 @@ set (SRCS
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64_stubs.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/X86Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
|
||||
Interface/Core/Interpreter/InterpreterFallbacks.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -137,14 +142,23 @@ set (SRCS
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/Allocator/64BitAllocator.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (_M_ARM_64)
|
||||
list(APPEND SRCS Utils/ArchHelpers/Arm64.cpp)
|
||||
else()
|
||||
list(APPEND SRCS Utils/ArchHelpers/Arm64_stubs.cpp)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
list(APPEND FEXCORE_BASE_SRCS
|
||||
Utils/AllocatorOverride.cpp)
|
||||
endif()
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
@@ -162,11 +176,6 @@ if (ENABLE_INTERPRETER)
|
||||
Interface/Core/Interpreter/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp)
|
||||
endif()
|
||||
|
||||
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
|
||||
|
||||
if (_M_X86_64)
|
||||
@@ -222,11 +231,22 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxhash tiny-json FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
else()
|
||||
list (APPEND LIBS synchronization)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
@@ -331,7 +351,7 @@ function(AddDefaultOptionsToTarget Name)
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
target_compile_definitions(${Name} PRIVATE ${DEFINES})
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
add_dependencies(${Name} CONFIG_INC IR_INC)
|
||||
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
@@ -373,7 +393,6 @@ AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
@@ -385,6 +404,16 @@ function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
if (MINGW_BUILD)
|
||||
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
|
||||
set_target_properties(${Name} PROPERTIES NO_SONAME ON)
|
||||
# Change the suffixes otherwise cmake continues using .a and .so
|
||||
if (${Type} STREQUAL SHARED)
|
||||
set_target_properties(${Name} PROPERTIES SUFFIX ".dll")
|
||||
elseif(${Type} STREQUAL STATIC)
|
||||
set_target_properties(${Name} PROPERTIES SUFFIX ".lib")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
endfunction()
|
||||
@@ -393,10 +422,12 @@ AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries)
|
||||
if (NOT MINGW_BUILD)
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
+8
-9
@@ -1,11 +1,10 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <string>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() {
|
||||
}
|
||||
@@ -18,7 +17,7 @@ namespace FEXCore {
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
@@ -28,7 +27,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fmt::format("{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
@@ -40,7 +39,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fmt::format("{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
@@ -52,7 +51,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fmt::format("{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
@@ -64,7 +63,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
@@ -76,7 +75,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
|
||||
-91
@@ -1,91 +0,0 @@
|
||||
#include "Common/Paths.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <pwd.h>
|
||||
#include <system_error>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
std::unique_ptr<std::string> CachePath;
|
||||
std::unique_ptr<std::string> EntryCache;
|
||||
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
return passwd->pw_dir;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const char *GetHomeDirectory() {
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
}
|
||||
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
return HomeDir;
|
||||
}
|
||||
|
||||
void InitializePaths() {
|
||||
CachePath = std::make_unique<std::string>();
|
||||
EntryCache = std::make_unique<std::string>();
|
||||
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
char *XDGDataDir = getenv("XDG_DATA_DIR");
|
||||
if (XDGDataDir) {
|
||||
*CachePath = XDGDataDir;
|
||||
}
|
||||
else {
|
||||
if (HomeDir) {
|
||||
*CachePath = HomeDir;
|
||||
}
|
||||
}
|
||||
|
||||
*CachePath += "/.fex-emu/";
|
||||
*EntryCache = *CachePath + "/EntryCache/";
|
||||
|
||||
std::error_code ec{};
|
||||
// Ensure the folder structure is created for our Data
|
||||
if (!std::filesystem::exists(*EntryCache, ec) &&
|
||||
!std::filesystem::create_directories(*EntryCache, ec)) {
|
||||
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
|
||||
}
|
||||
}
|
||||
|
||||
void ShutdownPaths() {
|
||||
CachePath.reset();
|
||||
EntryCache.reset();
|
||||
}
|
||||
|
||||
std::string GetCachePath() {
|
||||
return *CachePath;
|
||||
}
|
||||
|
||||
std::string GetEntryCachePath() {
|
||||
return *EntryCache;
|
||||
}
|
||||
}
|
||||
-12
@@ -1,12 +0,0 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
void InitializePaths();
|
||||
void ShutdownPaths();
|
||||
|
||||
const char *GetHomeDirectory();
|
||||
|
||||
std::string GetCachePath();
|
||||
std::string GetEntryCachePath();
|
||||
}
|
||||
+38
-23
@@ -2,19 +2,19 @@
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <sstream>
|
||||
|
||||
extern "C" {
|
||||
#include "SoftFloat-3e/platform.h"
|
||||
#include "SoftFloat-3e/softfloat.h"
|
||||
}
|
||||
|
||||
struct X80SoftFloat {
|
||||
struct FEX_PACKED X80SoftFloat {
|
||||
#ifdef _M_X86_64
|
||||
// Define this to push some operations to x87
|
||||
// Only useful to see if precision loss is killing something
|
||||
@@ -32,22 +32,28 @@ struct X80SoftFloat {
|
||||
#else
|
||||
#error No 128bit float for this target!
|
||||
#endif
|
||||
struct __attribute__((packed)) {
|
||||
uint64_t Significand : 64;
|
||||
uint16_t Exponent : 15;
|
||||
unsigned Sign : 1;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
#define LIBRARY_PRECISION BIGFLOAT
|
||||
#else
|
||||
// Mingw Win32 libraries don't have `__float128` helpers. Needs to use a lower precision.
|
||||
#define LIBRARY_PRECISION double
|
||||
#endif
|
||||
|
||||
uint64_t Significand : 64;
|
||||
uint16_t Exponent : 15;
|
||||
uint16_t Sign : 1;
|
||||
|
||||
X80SoftFloat() { memset(this, 0, sizeof(*this)); }
|
||||
X80SoftFloat(unsigned _Sign, uint16_t _Exponent, uint64_t _Significand)
|
||||
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
|
||||
: Significand {_Significand}
|
||||
, Exponent {_Exponent}
|
||||
, Sign {_Sign}
|
||||
{
|
||||
}
|
||||
|
||||
std::string str() const {
|
||||
std::ostringstream string;
|
||||
fextl::string str() const {
|
||||
fextl::ostringstream string;
|
||||
string << std::hex << Sign;
|
||||
string << "_" << Exponent;
|
||||
string << "_" << (Significand >> 63);
|
||||
@@ -262,7 +268,7 @@ struct X80SoftFloat {
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
|
||||
BIGFLOAT Src2_d = Int;
|
||||
LIBRARY_PRECISION Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
@@ -286,8 +292,8 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Result = exp2l(Src1_d);
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
#endif
|
||||
@@ -311,9 +317,9 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src2_d = rhs;
|
||||
LIBRARY_PRECISION Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
@@ -336,9 +342,9 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src2_d = rhs;
|
||||
LIBRARY_PRECISION Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
@@ -360,7 +366,7 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
@@ -382,7 +388,7 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
@@ -404,7 +410,7 @@ struct X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
@@ -439,6 +445,7 @@ struct X80SoftFloat {
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
operator BIGFLOAT() const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
@@ -449,6 +456,7 @@ struct X80SoftFloat {
|
||||
return result;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
operator int16_t() const {
|
||||
auto rv = extF80_to_i32(*this, softfloat_roundingMode, false);
|
||||
@@ -517,6 +525,7 @@ struct X80SoftFloat {
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
X80SoftFloat(BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
|
||||
@@ -524,6 +533,7 @@ struct X80SoftFloat {
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
X80SoftFloat(const int16_t rhs) {
|
||||
*this = i32_to_extF80(rhs);
|
||||
@@ -562,4 +572,9 @@ private:
|
||||
static constexpr uint32_t ExponentBias = 16383;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
static_assert(sizeof(X80SoftFloat) == 10, "tword must be 10bytes in size");
|
||||
#else
|
||||
// Padding on this extends to 16-bytes rather than 10-bytes on WIN32.
|
||||
static_assert(sizeof(X80SoftFloat) == 16, "tword must be 16bytes in size");
|
||||
#endif
|
||||
+13
-12
@@ -1,48 +1,49 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, bool *Result) {
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint8_t *Result) {
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint16_t *Result) {
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint32_t *Result) {
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, int32_t *Result) {
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint64_t *Result) {
|
||||
*Result = std::stoull(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, std::string *Result) {
|
||||
*Result = Value;
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
template <typename T,
|
||||
typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, T *Result) {
|
||||
*Result = static_cast<T>(std::stoull(std::string(Value), nullptr, 0));
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, fextl::string *Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
+10
-9
@@ -1,11 +1,11 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
namespace FEXCore::StringUtils {
|
||||
// Trim the left side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
|
||||
[[maybe_unused]] static fextl::string LeftTrim(fextl::string String, std::string_view TrimTokens = " \t\n\r") {
|
||||
size_t pos = fextl::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != fextl::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
@@ -13,9 +13,9 @@ namespace FEXCore::StringUtils {
|
||||
}
|
||||
|
||||
// Trim the right side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
|
||||
[[maybe_unused]] static fextl::string RightTrim(fextl::string String, std::string_view TrimTokens = " \t\n\r") {
|
||||
size_t pos = fextl::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != fextl::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
@@ -23,7 +23,8 @@ namespace FEXCore::StringUtils {
|
||||
}
|
||||
|
||||
// Trim both the left and right of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
|
||||
[[maybe_unused]] static fextl::string Trim(fextl::string String, std::string_view TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(std::move(String), TrimTokens), TrimTokens);
|
||||
}
|
||||
|
||||
}
|
||||
+121
-154
@@ -1,31 +1,31 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Common/Paths.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <list>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/sysinfo.h>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
@@ -45,13 +45,13 @@ namespace DefaultValues {
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
std::unique_ptr<std::list<json_t>> json_objects;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = std::make_unique<std::list<json_t>>();
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
@@ -60,8 +60,8 @@ namespace JSON {
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
std::vector<char> Data;
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
@@ -107,108 +107,75 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::string GetDataDirectory() {
|
||||
std::string DataDir{};
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_LAST,
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *DataXDG = getenv("XDG_DATA_HOME");
|
||||
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
if (DataOverride) {
|
||||
// Data override will override the complete directory
|
||||
DataDir = DataOverride;
|
||||
}
|
||||
else {
|
||||
DataDir = DataXDG ?: HomeDir;
|
||||
DataDir += "/.fex-emu/";
|
||||
}
|
||||
return DataDir;
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
}
|
||||
|
||||
std::string GetConfigDirectory(bool Global) {
|
||||
std::string ConfigDir;
|
||||
if (Global) {
|
||||
ConfigDir = GLOBAL_DATA_DIRECTORY;
|
||||
}
|
||||
else {
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
|
||||
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (ConfigOverride) {
|
||||
// Config override completely overrides the config directory
|
||||
ConfigDir = ConfigOverride;
|
||||
}
|
||||
else {
|
||||
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
|
||||
ConfigDir += "/.fex-emu/";
|
||||
}
|
||||
|
||||
// Ensure the folder structure is created for our configuration
|
||||
std::error_code ec{};
|
||||
if (!std::filesystem::exists(ConfigDir, ec) &&
|
||||
!std::filesystem::create_directories(ConfigDir, ec)) {
|
||||
// Let's go local in this case
|
||||
return "./";
|
||||
}
|
||||
}
|
||||
|
||||
return ConfigDir;
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
std::string GetConfigFileLocation(bool Global) {
|
||||
std::string ConfigFile{};
|
||||
if (Global) {
|
||||
ConfigFile = GetConfigDirectory(true) + "Config.json";
|
||||
}
|
||||
else {
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
}
|
||||
}
|
||||
return ConfigFile;
|
||||
void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
|
||||
std::string ConfigFile = GetConfigDirectory(Global);
|
||||
fextl::string const& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
}
|
||||
|
||||
fextl::string const& GetConfigDirectory(bool Global) {
|
||||
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
|
||||
}
|
||||
|
||||
fextl::string const& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
std::error_code ec{};
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
!FHU::Filesystem::Exists(ConfigFile) &&
|
||||
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
|
||||
// Let's go local in this case
|
||||
return "./" + Filename + ".json";
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
}
|
||||
|
||||
ConfigFile += "AppConfig/";
|
||||
|
||||
// Attempt to create the local folder if it doesn't exist
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
!FHU::Filesystem::Exists(ConfigFile) &&
|
||||
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
// Let's go local in this case
|
||||
return "./" + Filename + ".json";
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
}
|
||||
|
||||
ConfigFile += Filename + ".json";
|
||||
return ConfigFile;
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, std::string const &Config) {
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, fextl::string const &Config) {
|
||||
}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context *CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
|
||||
@@ -268,17 +235,17 @@ namespace JSON {
|
||||
}
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
std::unordered_map<std::string, std::string> LookupMap;
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](FEXCore::Config::LayerValue const &Value) {
|
||||
for (const auto &EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == std::string::npos) {
|
||||
if (ItEq == fextl::string::npos) {
|
||||
// Broken environment variable
|
||||
// Skip
|
||||
continue;
|
||||
}
|
||||
auto Key = std::string(EnvVar.begin(), EnvVar.begin() + ItEq);
|
||||
auto Value = std::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
|
||||
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
|
||||
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
|
||||
|
||||
// Add the key to the map, overwriting whatever previous value was there
|
||||
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
|
||||
@@ -311,7 +278,7 @@ namespace JSON {
|
||||
}
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(std::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
}
|
||||
|
||||
@@ -329,16 +296,15 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
|
||||
fextl::string ExpandPath(fextl::string const &ContainerPrefix, fextl::string PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
std::filesystem::path Path{PathName};
|
||||
|
||||
// Expand home if it exists
|
||||
if (Path.is_relative()) {
|
||||
std::string Home = getenv("HOME") ?: "";
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
@@ -347,12 +313,15 @@ namespace JSON {
|
||||
}
|
||||
|
||||
// Expand relative path to absolute
|
||||
Path = std::filesystem::absolute(Path);
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char *RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
}
|
||||
|
||||
// Only return if it exists
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(Path, ec)) {
|
||||
return Path;
|
||||
if (FHU::Filesystem::Exists(PathName)) {
|
||||
return PathName;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -368,9 +337,9 @@ namespace JSON {
|
||||
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
|
||||
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
|
||||
if (!ContainerPrefix.empty() && !PathName.empty()) {
|
||||
if (!std::filesystem::exists(PathName)) {
|
||||
if (!FHU::Filesystem::Exists(PathName)) {
|
||||
auto ContainerPath = ContainerPrefix + PathName;
|
||||
if (std::filesystem::exists(ContainerPath)) {
|
||||
if (FHU::Filesystem::Exists(ContainerPath)) {
|
||||
return ContainerPath;
|
||||
}
|
||||
}
|
||||
@@ -379,15 +348,15 @@ namespace JSON {
|
||||
return {};
|
||||
}
|
||||
|
||||
constexpr char ContainerManager[] = "/run/host/container-manager";
|
||||
|
||||
std::string FindContainer() {
|
||||
fextl::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
@@ -395,14 +364,13 @@ namespace JSON {
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string FindContainerPrefix() {
|
||||
fextl::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
@@ -424,7 +392,7 @@ namespace JSON {
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, std::to_string(get_nprocs_conf()));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -443,7 +411,7 @@ namespace JSON {
|
||||
#endif
|
||||
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -457,8 +425,8 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
fextl::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
@@ -467,16 +435,15 @@ namespace JSON {
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix,PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
std::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(NamedRootFS, ec)) {
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
@@ -498,9 +465,8 @@ namespace JSON {
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
std::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(NamedConfig, ec)) {
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
@@ -514,11 +480,11 @@ namespace JSON {
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, std::to_string(1u));
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
}
|
||||
}
|
||||
|
||||
void AddLayer(std::unique_ptr<FEXCore::Config::Layer> _Layer) {
|
||||
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
|
||||
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
|
||||
}
|
||||
|
||||
@@ -530,7 +496,7 @@ namespace JSON {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
std::optional<std::string*> Get(ConfigOption Option) {
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
@@ -571,7 +537,7 @@ namespace JSON {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string Default) {
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
@@ -582,13 +548,13 @@ namespace JSON {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
}
|
||||
else {
|
||||
return std::string(Default);
|
||||
return fextl::string(Default);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -603,39 +569,39 @@ namespace JSON {
|
||||
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
|
||||
|
||||
// Constructor
|
||||
template Value<std::string>::Value(FEXCore::Config::ConfigOption _Option, std::string Default);
|
||||
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
|
||||
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
|
||||
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
@@ -647,11 +613,11 @@ namespace JSON {
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
@@ -672,7 +638,7 @@ namespace JSON {
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
@@ -683,7 +649,7 @@ namespace JSON {
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type)
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
@@ -705,12 +671,13 @@ namespace JSON {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
std::unordered_map<std::string_view, std::string_view> EnvMap;
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (std::string::npos == pos)
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
@@ -719,10 +686,10 @@ namespace JSON {
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key]=Value;
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
@@ -739,31 +706,31 @@ namespace JSON {
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(it.first)).has_value()) {
|
||||
Set(it.second, std::string(*Value));
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -53,7 +53,7 @@
|
||||
},
|
||||
"EnableAVX": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Determines whether or not we use the expanded register file for AVX or not"
|
||||
]
|
||||
|
||||
+6
-53
@@ -1,4 +1,3 @@
|
||||
#include "Common/Paths.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
@@ -19,34 +18,18 @@ namespace FEXCore::HLE {
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
FEXCore::Paths::InitializePaths();
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
void ShutdownStaticTables() {
|
||||
FEXCore::Paths::ShutdownPaths();
|
||||
}
|
||||
|
||||
FEXCore::Context::Context *FEXCore::Context::Context::CreateNewContext() {
|
||||
return new FEXCore::Context::ContextImpl{};
|
||||
}
|
||||
|
||||
void FEXCore::Context::Context::DestroyContext(FEXCore::Context::Context *CTX) {
|
||||
CTX->DestroyContext();
|
||||
delete CTX;
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
return FEXCore::CPU::CreateCPUCore(this);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::DestroyContext() {
|
||||
if (ParentThread) {
|
||||
DestroyThread(ParentThread);
|
||||
}
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
@@ -104,41 +87,11 @@ namespace FEXCore::Context {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
|
||||
return CPUID.RunXCRFunction(Function);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
void SetVDSOSigReturn(FEXCore::Context::Context *CTX, const VDSOSigReturn &Pointers) {
|
||||
CTX->SetVDSOSigReturn(Pointers);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
//void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
// CTX->CompileRIP(CTX->ParentThread, RIP);
|
||||
//}
|
||||
//uint64_t GetThreadCount(FEXCore::Context::Context *CTX) {
|
||||
// return CTX->GetThreadCount();
|
||||
//}
|
||||
|
||||
//FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(FEXCore::Context::Context *CTX, uint64_t Thread) {
|
||||
// return CTX->GetRuntimeStatsForThread(Thread);
|
||||
//}
|
||||
|
||||
//bool GetDebugDataForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
// return CTX->GetDebugDataForRIP(RIP, Data);
|
||||
//}
|
||||
|
||||
//bool FindHostCodeForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, uint8_t **Code) {
|
||||
// return CTX->FindHostCodeForRIP(RIP, Code);
|
||||
//}
|
||||
|
||||
// XXX:
|
||||
// bool FindIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir) {
|
||||
// return CTX->FindIRForRIP(RIP, ir);
|
||||
// }
|
||||
|
||||
// void SetIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir) {
|
||||
// CTX->SetIRForRIP(RIP, ir);
|
||||
// }
|
||||
}
|
||||
|
||||
}
|
||||
+65
-47
@@ -1,7 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
@@ -14,24 +13,25 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <string>
|
||||
#include <unordered_map>
|
||||
#include <queue>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
@@ -75,8 +75,6 @@ namespace FEXCore::Context {
|
||||
// Context base class implementation.
|
||||
bool InitializeContext() override;
|
||||
|
||||
void DestroyContext() override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
@@ -108,9 +106,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) override;
|
||||
[[noreturn]] void HandleSignalHandlerReturn(bool RT) override ;
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) override;
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
@@ -161,37 +157,39 @@ namespace FEXCore::Context {
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string& Name) override;
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) override {
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) override {
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) override {
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) override {
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
|
||||
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
@@ -239,14 +237,13 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
} Config;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
FEXCore::Core::InternalThreadState* ParentThread{};
|
||||
std::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
@@ -262,19 +259,18 @@ namespace FEXCore::Context {
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
|
||||
SignalDelegator *SignalDelegation{};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
VDSOSigReturn VDSOPointers{};
|
||||
|
||||
ContextImpl();
|
||||
~ContextImpl();
|
||||
@@ -294,7 +290,8 @@ namespace FEXCore::Context {
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(static_cast<ContextImpl*>(Frame->Thread->CTX)->CodeInvalidationMutex);
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -306,19 +303,13 @@ namespace FEXCore::Context {
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
uint64_t GetThreadCount() const;
|
||||
FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(uint64_t Thread);
|
||||
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
@@ -354,31 +345,55 @@ namespace FEXCore::Context {
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void SetVDSOSigReturn(const VDSOSigReturn &Pointers) override {
|
||||
VDSOPointers = Pointers;
|
||||
if (VDSOPointers.VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers.VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
|
||||
if (VDSOPointers.VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers.VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
void IncrementIdleRefCount() override {
|
||||
++IdleWaitRefCount;
|
||||
}
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override { ExitOnHLT = true; }
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
}
|
||||
else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
@@ -406,17 +421,20 @@ namespace FEXCore::Context {
|
||||
|
||||
// Entry Cache
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
fextl::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
fextl::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
@@ -22,15 +23,35 @@ namespace FEXCore::CPU {
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : nullptr, size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
ConfiguredGPRs = NumGPRs64;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs64;
|
||||
ConfiguredGPRPairs = NumGPRPairs64;
|
||||
ConfiguredFPRs = NumFPRs64;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs64;
|
||||
ConfiguredDynamicGPRs = NumGPRs64 - NumGPRs64; // Will be zero, just to be consistent with 32-bit side
|
||||
ConfiguredDynamicRegisterBase = nullptr;
|
||||
}
|
||||
else {
|
||||
ConfiguredGPRs = NumGPRs32;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs32;
|
||||
ConfiguredGPRPairs = NumGPRPairs32;
|
||||
ConfiguredFPRs = NumFPRs32;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs32;
|
||||
ConfiguredDynamicGPRs = NumGPRs32 - NumGPRs64; // Will be 8
|
||||
ConfiguredDynamicRegisterBase = &RA64[9];
|
||||
}
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::munmap(GetBufferBase(), BufferSize);
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -112,7 +133,11 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
const fextl::vector<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>> CalleeSaved = {{
|
||||
#ifdef _WIN32
|
||||
// Platform register, Just save it twice to make logic easy.
|
||||
{ARMEmitter::XReg::x18, ARMEmitter::XReg::x18},
|
||||
#endif
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
|
||||
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
|
||||
@@ -176,13 +201,17 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
32);
|
||||
}
|
||||
|
||||
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
const fextl::vector<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>> CalleeSaved = {{
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
{ARMEmitter::XReg::x27, ARMEmitter::XReg::x28},
|
||||
{ARMEmitter::XReg::x25, ARMEmitter::XReg::x26},
|
||||
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
|
||||
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
#ifdef _WIN32
|
||||
// Platform register.
|
||||
{ARMEmitter::XReg::x18, ARMEmitter::XReg::zr},
|
||||
#endif
|
||||
}};
|
||||
|
||||
for (auto &RegPair : CalleeSaved) {
|
||||
@@ -190,12 +219,12 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
@@ -212,7 +241,7 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
@@ -223,11 +252,9 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
} else {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRSpillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
@@ -236,7 +263,7 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
@@ -270,7 +297,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
@@ -285,7 +312,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
@@ -294,7 +321,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
@@ -313,7 +340,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
@@ -331,10 +358,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = 1 * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = (ConfiguredDynamicGPRs + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const auto FPRSize = ConfiguredFPRs * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -343,7 +370,7 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
@@ -352,8 +379,8 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
static_assert(RAFPR.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
LOGMAN_THROW_AA_FMT(ConfiguredFPRs % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
@@ -362,6 +389,14 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
}
|
||||
|
||||
@@ -369,7 +404,7 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
@@ -378,7 +413,7 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
@@ -387,6 +422,14 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
}
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
|
||||
+103
-15
@@ -28,35 +28,83 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// All but x29 are caller saved
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9, FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r18, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r15, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r13, FEXCore::ARMEmitter::Reg::r29
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
// All are callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r19
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9 + 8> RA64 = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RA64Pair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4 + 3> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
// Registers only available on 32-bit
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17}
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
/*FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,*/FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, // FEXCore::ARMEmitter::VReg::v0 ~ FEXCore::ARMEmitter::VReg::v3 are used as temps
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12 + 8> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
@@ -90,16 +138,56 @@ protected:
|
||||
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
uint32_t ConfiguredGPRs;
|
||||
uint32_t ConfiguredSRAGPRs;
|
||||
uint32_t ConfiguredGPRPairs;
|
||||
uint32_t ConfiguredFPRs;
|
||||
uint32_t ConfiguredSRAFPRs;
|
||||
uint32_t ConfiguredDynamicGPRs;
|
||||
const FEXCore::ARMEmitter::Register *ConfiguredDynamicRegisterBase{};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
// 64-bit gets removal of additional pairs
|
||||
constexpr static uint32_t NumGPRs64 = RA64.size() - 8;
|
||||
constexpr static uint32_t NumSRAGPRs64 = SRA64.size();
|
||||
constexpr static uint32_t NumFPRs64 = RAFPR.size() - 8;
|
||||
constexpr static uint32_t NumSRAFPRs64 = SRAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs64 = RA64Pair.size() - 3;
|
||||
|
||||
// 32-bit gets full array of GPR registers
|
||||
// SRA registers remove the additional 8
|
||||
constexpr static uint32_t NumGPRs32 = RA64.size();
|
||||
constexpr static uint32_t NumSRAGPRs32 = SRA64.size() - 8;
|
||||
constexpr static uint32_t NumFPRs32 = RAFPR.size();
|
||||
constexpr static uint32_t NumSRAFPRs32 = SRAFPR.size() - 8;
|
||||
constexpr static uint32_t NumGPRPairs32 = RA64Pair.size();
|
||||
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
constexpr static uint64_t FPRBase = (1ULL << 32);
|
||||
constexpr static uint64_t GPRPairBase = (2ULL << 32);
|
||||
|
||||
/** @} */
|
||||
|
||||
constexpr static uint8_t RA_32 = 0;
|
||||
constexpr static uint8_t RA_64 = 1;
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
void LoadConstant(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0110'0000'0000'0111'1111'1111'1111'1111U;
|
||||
|
||||
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
|
||||
// We can't guarantee only the lower 64bits are used so flush everything
|
||||
|
||||
+94
-190
@@ -1893,433 +1893,351 @@ public:
|
||||
// TODO: Double check narrowing op size limits.
|
||||
// TODO: Don't enforce DRegister/QRegister for Q check
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void saddl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void saddl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void saddl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void saddl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void saddw(SubRegSize size, T rd, T rn, T rm) {
|
||||
void saddw(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void saddw2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void saddw2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void ssubl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void ssubl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void ssubl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void ssubl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void ssubw(SubRegSize size, T rd, T rn, T rm) {
|
||||
void ssubw(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void ssubw2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void ssubw2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void addhn(SubRegSize size, T rd, T rn, T rm) {
|
||||
void addhn(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i64Bit, "No 64-bit dest support.");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
ASIMD3Different<T>(Op, 0, 0b0100, size, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0100, size, rd, rn, rm);
|
||||
}
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void addhn2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void addhn2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i64Bit, "No 64-bit dest support.");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
ASIMD3Different<T>(Op, 0, 0b0100, size, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0100, size, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void sabal(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sabal(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void sabal2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sabal2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void subhn(SubRegSize size, T rd, T rn, T rm) {
|
||||
void subhn(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i64Bit, "No 64-bit dest support.");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
ASIMD3Different<T>(Op, 0, 0b0110, size, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0110, size, rd, rn, rm);
|
||||
}
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void subhn2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void subhn2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i64Bit, "No 64-bit dest support.");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
ASIMD3Different<T>(Op, 0, 0b0110, size, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0110, size, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void sabdl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sabdl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0111, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0111, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void sabdl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sabdl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b0111, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b0111, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void smlal(SubRegSize size, T rd, T rn, T rm) {
|
||||
void smlal(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void smlal2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void smlal2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void sqdmlal(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sqdmlal(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i16Bit, "No 8/16-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1001, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1001, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void sqdmlal2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sqdmlal2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i16Bit, "No 8/16-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1001, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1001, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void smlsl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void smlsl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void smlsl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void smlsl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void sqdmlsl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sqdmlsl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i16Bit, "No 8/16-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1011, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1011, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void sqdmlsl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sqdmlsl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i16Bit, "No 8/16-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1011, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1011, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void smull(SubRegSize size, T rd, T rn, T rm) {
|
||||
void smull(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void smull2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void smull2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void sqdmull(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sqdmull(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i16Bit, "No 8/16-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1101, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1101, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void sqdmull2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void sqdmull2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i16Bit, "No 8/16-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 0, 0b1101, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1101, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void pmull(SubRegSize size, T rd, T rn, T rm) {
|
||||
void pmull(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i128Bit, "Only 16-bit and 128-bit destination supported");
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
ASIMD3Different<T>(Op, 0, 0b1110, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1110, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void pmull2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void pmull2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i128Bit, "Only 16-bit and 128-bit destination supported");
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
ASIMD3Different<T>(Op, 0, 0b1110, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 0, 0b1110, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void uaddl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uaddl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void uaddl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uaddl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void uaddw(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uaddw(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void uaddw2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uaddw2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0001, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void usubl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void usubl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void usubl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void usubl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void usubw(SubRegSize size, T rd, T rn, T rm) {
|
||||
void usubw(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void usubw2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void usubw2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0011, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
// XXX: RADDHN/2
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void uabal(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uabal(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void uabal2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uabal2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b0101, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
// XXX: RSUBHN/2
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void uabdl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uabdl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
@@ -2328,9 +2246,7 @@ public:
|
||||
ASIMD3Different(Op, 1, 0b0111, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void uabdl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void uabdl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
@@ -2339,70 +2255,58 @@ public:
|
||||
ASIMD3Different(Op, 1, 0b0111, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void umlal(SubRegSize size, T rd, T rn, T rm) {
|
||||
void umlal(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void umlal2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void umlal2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b1000, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void umlsl(SubRegSize size, T rd, T rn, T rm) {
|
||||
void umlsl(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void umlsl2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void umlsl2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b1010, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<DRegister, T>)
|
||||
void umull(SubRegSize size, T rd, T rn, T rm) {
|
||||
void umull(SubRegSize size, DRegister rd, DRegister rn, DRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
///< Size is dest size
|
||||
template<typename T>
|
||||
requires(std::is_same_v<QRegister, T>)
|
||||
void umull2(SubRegSize size, T rd, T rn, T rm) {
|
||||
void umull2(SubRegSize size, QRegister rd, QRegister rn, QRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size != SubRegSize::i8Bit, "No 8-bit dest support.");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'00 << 10;
|
||||
const auto ConvertedSize = SubRegSize{FEXCore::ToUnderlying(size) - 1};
|
||||
|
||||
ASIMD3Different<T>(Op, 1, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
ASIMD3Different(Op, 1, 0b1100, ConvertedSize, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Advanced SIMD three same
|
||||
|
||||
@@ -6,13 +6,13 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
/*
|
||||
* Welcome to FEX-Emu's custom AArch64 emitter.
|
||||
@@ -62,20 +62,9 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to get the `Size` enum in bits.
|
||||
template<Size size>
|
||||
constexpr size_t RegSizeInBits() {
|
||||
constexpr size_t RegSize[] = {
|
||||
32, 64, 128,
|
||||
};
|
||||
return RegSize[FEXCore::ToUnderlying(size)];
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t RegSizeInBits(Size size) {
|
||||
constexpr size_t RegSize[] = {
|
||||
32, 64, 128,
|
||||
};
|
||||
return RegSize[FEXCore::ToUnderlying(size)];
|
||||
[[nodiscard]]
|
||||
constexpr size_t RegSizeInBits(Size size) {
|
||||
return size_t{32} << FEXCore::ToUnderlying(size);
|
||||
}
|
||||
|
||||
/* This `SubRegSize` enum is used for most ASIMD operations.
|
||||
@@ -90,14 +79,9 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to get the `SubRegSize` in bits.
|
||||
template<SubRegSize size>
|
||||
constexpr size_t SubRegSizeInBits() {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t SubRegSizeInBits(SubRegSize size) {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
[[nodiscard]]
|
||||
constexpr size_t SubRegSizeInBits(SubRegSize size) {
|
||||
return size_t{8} << FEXCore::ToUnderlying(size);
|
||||
}
|
||||
|
||||
/* This `ScalarRegSize` enum is used for most scalar float
|
||||
@@ -117,14 +101,9 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to get the `ScalarRegSize` in bits.
|
||||
template<ScalarRegSize size>
|
||||
constexpr size_t ScalarRegSizeInBits() {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t ScalarRegSizeInBits(ScalarRegSize size) {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
[[nodiscard]]
|
||||
constexpr size_t ScalarRegSizeInBits(ScalarRegSize size) {
|
||||
return size_t{8} << FEXCore::ToUnderlying(size);
|
||||
}
|
||||
|
||||
/* This `VectorRegSizePair` union allows us to have an overlapping type
|
||||
@@ -140,12 +119,12 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to create a `VectorRegSizePair` union.
|
||||
[[maybe_unused]]
|
||||
static inline VectorRegSizePair ToVectorSizePair(SubRegSize size) {
|
||||
[[nodiscard]]
|
||||
constexpr VectorRegSizePair ToVectorSizePair(SubRegSize size) {
|
||||
return VectorRegSizePair {.Vector = size};
|
||||
}
|
||||
[[maybe_unused]]
|
||||
static inline VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
|
||||
[[nodiscard]]
|
||||
constexpr VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
|
||||
return VectorRegSizePair {.Scalar = size};
|
||||
}
|
||||
|
||||
@@ -524,7 +503,7 @@ namespace FEXCore::ARMEmitter {
|
||||
uint8_t *Location{};
|
||||
InstType Type;
|
||||
};
|
||||
std::vector<Instructions> Insts{};
|
||||
fextl::vector<Instructions> Insts{};
|
||||
};
|
||||
|
||||
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
|
||||
+916
-362
File diff suppressed because it is too large.
Load diff
+672
-1073
File diff suppressed because it is too large.
Load diff
+3
-2
@@ -1,3 +1,4 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
@@ -57,7 +58,7 @@ auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
@@ -67,7 +68,7 @@ auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
|
||||
+53
-43
@@ -8,18 +8,20 @@ $end_info$
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
|
||||
#include <cstring>
|
||||
#ifdef _M_X86_64
|
||||
#include <cpuid.h>
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -74,20 +76,6 @@ static uint32_t GetCPUID() {
|
||||
return CPU;
|
||||
}
|
||||
|
||||
static uint32_t CalculateNumberOfCPUs() {
|
||||
size_t CPUs = 1;
|
||||
|
||||
while(std::filesystem::exists("/sys/devices/system/cpu/cpu" + std::to_string(CPUs))) {
|
||||
CPUs++;
|
||||
}
|
||||
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
0 | // Stepping
|
||||
@@ -115,13 +103,13 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
|
||||
uint64_t MIDR{};
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
std::error_code ec{};
|
||||
std::string MIDRPath = fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
std::array<char, 18> Data;
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
@@ -217,8 +205,8 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
if (Hybrid) {
|
||||
// Walk the MIDRs and calculate big little designs
|
||||
std::vector<const CPUMIDR*> BigCores;
|
||||
std::vector<const CPUMIDR*> LittleCores;
|
||||
fextl::vector<const CPUMIDR*> BigCores;
|
||||
fextl::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
@@ -354,28 +342,28 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
#else
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
__cpuid(0, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x15) {
|
||||
__cpuid(0x15, eax, ebx, ecx, edx);
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0, data);
|
||||
if (data[0] >= 0x15) {
|
||||
Xbyak::util::Cpu::getCpuid(0x15, data);
|
||||
|
||||
if (eax && ebx && ecx) {
|
||||
return ecx * ebx / eax;
|
||||
if (data[0] && data[1] && data[2]) {
|
||||
return data[2] * data[1] / data[0];
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
__cpuid(0, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x7) {
|
||||
__cpuid(0x7, eax, ebx, ecx, edx);
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0, data);
|
||||
if (data[0] >= 0x7) {
|
||||
Xbyak::util::Cpu::getCpuid(0x7, data);
|
||||
// Bit 15 of edx claims hybrid CPU
|
||||
Hybrid = (edx & (1U << 15)) != 0;
|
||||
Hybrid = (data[3] & (1U << 15)) != 0;
|
||||
}
|
||||
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
PerCPUData[i].IsBig = true;
|
||||
@@ -449,7 +437,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
@@ -638,12 +626,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(1 << 3) | // BMI1
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(1 << 8) | // BMI2
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
@@ -707,7 +695,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 1) | // Reserved
|
||||
(0 << 2) | // AVX512_4VNNIW
|
||||
(0 << 3) | // AVX512_4FMAPS
|
||||
(0 << 4) | // Fast Short Rep Mov
|
||||
(1 << 4) | // Fast Short Rep Mov
|
||||
(0 << 5) | // Reserved
|
||||
(0 << 6) | // Reserved
|
||||
(0 << 7) | // Reserved
|
||||
@@ -744,13 +732,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
// Leaf 0
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
uint32_t XFeatureSupportedSizeMax = SUPPORTS_AVX ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
uint32_t XFeatureSupportedSizeMax = SupportsAVX() ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
if (Leaf == 0) {
|
||||
// XFeatureSupportedMask[31:0]
|
||||
Res.eax =
|
||||
(1 << 0) | // X87 support
|
||||
(1 << 1) | // 128-bit SSE support
|
||||
(SUPPORTS_AVX << 2) | // 256-bit AVX support
|
||||
(SupportsAVX() << 2) | // 256-bit AVX support
|
||||
(0b00 << 3) | // MPX State
|
||||
(0b000 << 5) | // AVX-512 state
|
||||
(0 << 8) | // "Used for IA32_XSS" ... Used for what?
|
||||
@@ -784,8 +772,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
Res.edx = 0;
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
Res.eax = SUPPORTS_AVX ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SUPPORTS_AVX ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
Res.eax = SupportsAVX() ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SupportsAVX() ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
|
||||
// Reserved
|
||||
Res.ecx = 0;
|
||||
@@ -880,6 +868,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
|
||||
// Extended processor and feature bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
|
||||
// RDTSCP is disabled on WIN32/Wine because there is no sane way to query processor ID.
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 0;
|
||||
#else
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 1;
|
||||
#endif
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
@@ -946,7 +941,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(1 << 27) | // RDTSCP
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(1 << 30) | // 3DNow! Extensions
|
||||
@@ -977,14 +972,14 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(16L, DESCRIBE_STR_SIZE));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(ssize_t{16L}, DESCRIBE_STR_SIZE));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(0L, DESCRIBE_STR_SIZE - 16));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(ssize_t{0L}, DESCRIBE_STR_SIZE - 16));
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -1213,11 +1208,26 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() {
|
||||
// This just returns XCR0
|
||||
FEXCore::CPUID::XCRResults Res{
|
||||
.eax = static_cast<uint32_t>(XCR0),
|
||||
.edx = static_cast<uint32_t>(XCR0 >> 32),
|
||||
};
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
CTX = ctx;
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+46
-5
@@ -1,12 +1,12 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -63,13 +63,52 @@ public:
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) {
|
||||
if (Function >= 1) {
|
||||
// XCR function 1 is not yet supported.
|
||||
return {};
|
||||
}
|
||||
|
||||
return XCRFunction_0h();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
// Affects XSAVE and XRSTOR when modified.
|
||||
// Bit layout is as follows.
|
||||
// [0] - x87 enabled
|
||||
// [1] - SSE enabled
|
||||
// [2] - YMM enabled (256-bit SSE)
|
||||
// [8:3] - Reserved. MBZ.
|
||||
// [9] - MPK
|
||||
// [10] - Reserved. MBZ.
|
||||
// [11] - CET_U
|
||||
// [12] - CET_S
|
||||
// [61:13] - Reserved. MBZ.
|
||||
// [62] - LWP (Lightweight profiling)
|
||||
// [63] - Reserved for XCR bit vector expansion. MBZ.
|
||||
// Always enable x87 and SSE by default.
|
||||
constexpr static uint64_t XCR0_X87 = 1ULL << 0;
|
||||
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
|
||||
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
|
||||
|
||||
uint64_t XCR0 {
|
||||
XCR0_X87 |
|
||||
XCR0_SSE
|
||||
};
|
||||
|
||||
uint32_t SupportsAVX() const {
|
||||
return (XCR0 & XCR0_AVX) ? 1 : 0;
|
||||
}
|
||||
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
@@ -77,7 +116,7 @@ private:
|
||||
#endif
|
||||
bool IsBig{};
|
||||
};
|
||||
std::vector<CPUData> PerCPUData{};
|
||||
fextl::vector<CPUData> PerCPUData{};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
@@ -109,6 +148,8 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h();
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
static constexpr std::array<FunctionHandler, 27> Primary = {
|
||||
// 0: Highest function parameter and ID
|
||||
|
||||
+128
-171
@@ -19,10 +19,12 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterCore.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
@@ -32,7 +34,6 @@ $end_info$
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
@@ -42,9 +43,15 @@ $end_info$
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
@@ -53,28 +60,19 @@ $end_info$
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <filesystem>
|
||||
#include <fcntl.h>
|
||||
#include <functional>
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <set>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
@@ -153,12 +151,8 @@ namespace FEXCore::Context {
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = std::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
if (!Config.EnableAVX) {
|
||||
HostFeatures.SupportsAVX = false;
|
||||
}
|
||||
|
||||
if (!Config.Is64BitMode()) {
|
||||
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
|
||||
Config.VirtualMemSize = 1ULL << 32;
|
||||
@@ -170,9 +164,16 @@ namespace FEXCore::Context {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
if (ParentThread) {
|
||||
DestroyThread(ParentThread);
|
||||
}
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
CodeObjectCacheService->Shutdown();
|
||||
@@ -191,27 +192,25 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
static FEXCore::Core::CPUState CreateDefaultCPUState() {
|
||||
FEXCore::Core::CPUState NewThreadState{};
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
const CPU::CPUBackend::JITCodeHeader *InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
|
||||
// Initialize default CPU state
|
||||
NewThreadState.rip = ~0ULL;
|
||||
for (auto& greg : NewThreadState.gregs) {
|
||||
greg = 0;
|
||||
if (InlineHeader) {
|
||||
const CPU::CPUBackend::JITCodeTail *InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
|
||||
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
return InlineTail->RIP;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto& xmm : NewThreadState.xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(NewThreadState.flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
NewThreadState.flags[1] = 1;
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
NewThreadState.FTW = 0xFFFF;
|
||||
return NewThreadState;
|
||||
// Fallback to what is stored in the RIP currently.
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
@@ -219,16 +218,13 @@ namespace FEXCore::Context {
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
FEXCore::CPU::InitializeInterpreterSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetInterpreterBackendFeatures();
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
@@ -252,24 +248,37 @@ namespace FEXCore::Context {
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
|
||||
#endif
|
||||
|
||||
// Initialize common signal handlers
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.StaticRegisterAllocation = DispatcherConfig.StaticRegisterAllocation,
|
||||
.SupportsAVX = HostFeatures.SupportsAVX,
|
||||
|
||||
auto PauseHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
|
||||
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
|
||||
|
||||
.PauseReturnInstruction = Dispatcher->PauseReturnInstruction,
|
||||
.ThreadPauseHandlerAddressSpillSRA = Dispatcher->ThreadPauseHandlerAddressSpillSRA,
|
||||
.ThreadPauseHandlerAddress = Dispatcher->ThreadPauseHandlerAddress,
|
||||
|
||||
// Stop handlers.
|
||||
.ThreadStopHandlerAddressSpillSRA = Dispatcher->ThreadStopHandlerAddressSpillSRA,
|
||||
.ThreadStopHandlerAddress = Dispatcher->ThreadStopHandlerAddress,
|
||||
|
||||
// SRA information.
|
||||
.SRAGPRCount = Dispatcher->GetSRAGPRCount(),
|
||||
.SRAFPRCount = Dispatcher->GetSRAFPRCount(),
|
||||
};
|
||||
|
||||
SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, PauseHandler, true);
|
||||
Dispatcher->GetSRAGPRMapping(SignalConfig.SRAGPRMapping);
|
||||
Dispatcher->GetSRAFPRMapping(SignalConfig.SRAFPRMapping);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
return static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->HandleGuestSignal(Thread, Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
|
||||
// Initialize GDBServer after the signal handlers are installed
|
||||
// It may install its own handlers that need to be executed AFTER the CPU cores
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
@@ -277,12 +286,13 @@ namespace FEXCore::Context {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
ThunkHandler.reset(FEXCore::ThunkHandler::Create());
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0);
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
@@ -296,43 +306,24 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::StartGdbServer() {
|
||||
#ifndef _WIN32
|
||||
if (!DebugServer) {
|
||||
DebugServer = std::make_unique<GdbServer>(this);
|
||||
DebugServer = fextl::make_unique<GdbServer>(this);
|
||||
StartPaused = true;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::StopGdbServer() {
|
||||
#ifndef _WIN32
|
||||
DebugServer.reset();
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
void ContextImpl::HandleSignalHandlerReturn(bool RT) {
|
||||
using SignalHandlerReturnFunc = void(*)();
|
||||
|
||||
SignalHandlerReturnFunc SignalHandlerReturn{};
|
||||
if (RT) {
|
||||
SignalHandlerReturn = reinterpret_cast<SignalHandlerReturnFunc>(Dispatcher->SignalHandlerReturnAddressRT);
|
||||
}
|
||||
else {
|
||||
SignalHandlerReturn = reinterpret_cast<SignalHandlerReturnFunc>(Dispatcher->SignalHandlerReturnAddress);
|
||||
}
|
||||
|
||||
SignalHandlerReturn();
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
void ContextImpl::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void ContextImpl::RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SignalDelegation->RegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForIdle() {
|
||||
std::unique_lock<std::mutex> lk(IdleWaitMutex);
|
||||
IdleWaitCV.wait(lk, [this] {
|
||||
@@ -365,11 +356,7 @@ namespace FEXCore::Context {
|
||||
// Tell all the threads that they should pause
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Pause);
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
// Only attempt to stop this thread if it is running
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Pause);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -470,15 +457,13 @@ namespace FEXCore::Context {
|
||||
|
||||
void ContextImpl::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->RunningEvents.Running.exchange(false)) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Stop);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Stop);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
Thread->SignalReason.store(Event);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
SignalDelegation->SignalThread(Thread, Event);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -550,7 +535,9 @@ namespace FEXCore::Context {
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -559,11 +546,11 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
Thread->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
Thread->PassManager = std::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->PassManager->RegisterExitHandler([this]() {
|
||||
Stop(false /* Ignore current thread */);
|
||||
});
|
||||
@@ -614,7 +601,9 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
@@ -623,6 +612,9 @@ namespace FEXCore::Context {
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
|
||||
// Insert after the Thread object has been fully initialized
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
@@ -648,6 +640,8 @@ namespace FEXCore::Context {
|
||||
// To be able to delete a thread from itself, we need to detached the std::thread object
|
||||
Thread->ExecutionThread->detach();
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
@@ -710,49 +704,46 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
FEXCore::File::File FD;
|
||||
const auto DumpIRStr = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR();
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpIRStr =="stderr" || DumpIRStr =="no") {
|
||||
f = stderr;
|
||||
FD = FEXCore::File::File::GetStdERR();
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
f = stdout;
|
||||
FD = FEXCore::File::File::GetStdOUT();
|
||||
}
|
||||
else {
|
||||
const auto fileName = fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
f = fopen(fileName.c_str(), "w");
|
||||
CloseAfter = true;
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
}
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
if (FD.IsValid()) {
|
||||
fextl::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fmt::print(f,"IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
}
|
||||
};
|
||||
|
||||
static void ValidateIR(ContextImpl *ctx, IR::IREmitter *IREmitter) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
fextl::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction(ctx->OpDispatcherAllocator);
|
||||
compaction->Run(IREmitter);
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
FEXCore::Utils::PooledAllocatorMalloc Allocator;
|
||||
auto reparsed = IR::Parse(Allocator, &out);
|
||||
auto reparsed = IR::Parse(Allocator, out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
std::stringstream out2;
|
||||
fextl::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
@@ -790,7 +781,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
}
|
||||
});
|
||||
|
||||
@@ -864,6 +855,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
@@ -966,7 +960,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
@@ -975,7 +969,7 @@ namespace FEXCore::Context {
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -998,9 +992,6 @@ namespace FEXCore::Context {
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
// Increment stats
|
||||
Thread->Stats.BlocksCompiled.fetch_add(1);
|
||||
|
||||
// These blocks aren't already in the cache
|
||||
GeneratedIR = true;
|
||||
}
|
||||
@@ -1071,7 +1062,7 @@ namespace FEXCore::Context {
|
||||
auto FragmentBasePtr = reinterpret_cast<uint8_t *>(CodePtr);
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
@@ -1096,7 +1087,7 @@ namespace FEXCore::Context {
|
||||
if (CodeObjectCacheService &&
|
||||
Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE &&
|
||||
DebugData) {
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(std::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(fextl::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(
|
||||
CodeSerialize::AsyncJobHandler::SerializationJobData {
|
||||
.GuestRIP = GuestRIP,
|
||||
.GuestCodeLength = Length,
|
||||
@@ -1139,6 +1130,7 @@ namespace FEXCore::Context {
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
@@ -1183,6 +1175,7 @@ namespace FEXCore::Context {
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
@@ -1213,14 +1206,20 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex);
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex);
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1229,6 +1228,7 @@ namespace FEXCore::Context {
|
||||
void ContextImpl::MarkMemoryShared() {
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
@@ -1282,63 +1282,18 @@ namespace FEXCore::Context {
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidateGuestCodeRange(Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
InvalidateGuestCodeRange(nullptr, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
});
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
void Context::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
uint64_t RIPBackup = Thread->CurrentFrame->State.rip;
|
||||
Thread->CurrentFrame->State.rip = RIP;
|
||||
|
||||
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
CTX->ThreadRemoveCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CTX->CompileBlock(Thread->CurrentFrame, RIP);
|
||||
|
||||
Thread->CurrentFrame->State.rip = RIPBackup;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::GetThreadCount() const {
|
||||
return Threads.size();
|
||||
}
|
||||
|
||||
FEXCore::Core::RuntimeStats *ContextImpl::GetRuntimeStatsForThread(uint64_t Thread) {
|
||||
return &Threads[Thread]->Stats;
|
||||
}
|
||||
|
||||
bool ContextImpl::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
std::lock_guard<std::recursive_mutex> lk(ParentThread->LookupCache->WriteLock);
|
||||
auto it = ParentThread->DebugStore.find(RIP);
|
||||
if (it == ParentThread->DebugStore.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(Data, it->second.DebugData.get(), sizeof(FEXCore::Core::DebugData));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ContextImpl::FindHostCodeForRIP(uint64_t RIP, uint8_t **Code) {
|
||||
uintptr_t HostCode = ParentThread->LookupCache->FindBlock(RIP);
|
||||
if (!HostCode) {
|
||||
return false;
|
||||
}
|
||||
|
||||
*Code = reinterpret_cast<uint8_t*>(HostCode);
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) {
|
||||
uint64_t Result{};
|
||||
Result = Handler->HandleSyscall(Frame, Args);
|
||||
return Result;
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const std::string &filename) {
|
||||
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
@@ -1353,11 +1308,13 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
|
||||
@@ -12,6 +11,7 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <array>
|
||||
@@ -28,7 +28,6 @@
|
||||
#include <code-buffer-vixl.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -130,13 +129,6 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
@@ -184,7 +176,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -198,26 +190,11 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
@@ -229,26 +206,17 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(ARMEmitter::XReg::x4, ARMEmitter::XReg::x0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
mov(ARMEmitter::XReg::x0, ARMEmitter::XReg::x4);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
|
||||
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
@@ -257,29 +225,11 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
Bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, 8);
|
||||
}
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
@@ -292,23 +242,17 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP1, 0);
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
@@ -334,7 +278,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
@@ -345,7 +289,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
brk(0);
|
||||
}
|
||||
@@ -356,20 +300,28 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
if (CTX->ExitOnHLTEnabled()) {
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
|
||||
PopCalleeSavedRegisters();
|
||||
ret();
|
||||
}
|
||||
else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
@@ -446,7 +398,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -468,7 +420,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -490,7 +442,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -512,7 +464,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
@@ -543,7 +495,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
ClearICache(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
fextl::string Name = fextl::fmt::format("Dispatch_{}", FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
@@ -626,28 +578,6 @@ size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].Idx())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].Idx());
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
@@ -672,8 +602,8 @@ void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thr
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return fextl::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -30,8 +30,25 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
|
||||
void EmitDispatcher();
|
||||
|
||||
protected:
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
uint16_t GetSRAGPRCount() const override {
|
||||
return SRA64.size();
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const override {
|
||||
return SRAFPR.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRA64.size(); ++i) {
|
||||
Mapping[i] = SRA64[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRAFPR.size(); ++i) {
|
||||
Mapping[i] = SRAFPR[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Long division helpers
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
+19
-80
@@ -1,14 +1,13 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestSigAction;
|
||||
@@ -57,14 +56,6 @@ public:
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
// These are across all arches for now
|
||||
@@ -74,8 +65,8 @@ public:
|
||||
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
|
||||
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
static fextl::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
static fextl::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
|
||||
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
@@ -85,80 +76,28 @@ public:
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
virtual uint16_t GetSRAGPRCount() const {
|
||||
return 0U;
|
||||
}
|
||||
|
||||
virtual uint16_t GetSRAFPRCount() const {
|
||||
return 0U;
|
||||
}
|
||||
|
||||
virtual void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
}
|
||||
|
||||
virtual void GetSRAFPRMapping(uint8_t Mapping[16]) const {
|
||||
}
|
||||
|
||||
const DispatcherConfig& GetConfig() const { return config; }
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
, config {Config}
|
||||
{}
|
||||
|
||||
uint64_t ReconstructRIPFromContext(FEXCore::Core::CpuStateFrame *Frame, void *ucontext) const;
|
||||
void RestoreFrame_x64(ArchHelpers::Context::ContextBackup* Context, FEXCore::Core::CpuStateFrame *Frame, void *ucontext);
|
||||
void RestoreFrame_ia32(ArchHelpers::Context::ContextBackup* Context, FEXCore::Core::CpuStateFrame *Frame, void *ucontext);
|
||||
void RestoreRTFrame_ia32(ArchHelpers::Context::ContextBackup* Context, FEXCore::Core::CpuStateFrame *Frame, void *ucontext);
|
||||
|
||||
///< Setup the signal frame for x64.
|
||||
uint64_t SetupFrame_x64(FEXCore::Core::InternalThreadState *Thread, ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame *Frame,
|
||||
int Signal, siginfo_t *HostSigInfo, void *ucontext,
|
||||
GuestSigAction *GuestAction, stack_t *GuestStack,
|
||||
uint64_t NewGuestSP, const uint32_t eflags);
|
||||
|
||||
///< Setup the signal frame for a 32-bit signal without SA_SIGINFO.
|
||||
uint64_t SetupFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame *Frame,
|
||||
int Signal, siginfo_t *HostSigInfo, void *ucontext,
|
||||
GuestSigAction *GuestAction, stack_t *GuestStack,
|
||||
uint64_t NewGuestSP, const uint32_t eflags);
|
||||
|
||||
///< Setup the signal frame for a 32-bit signal with SA_SIGINFO.
|
||||
uint64_t SetupRTFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame *Frame,
|
||||
int Signal, siginfo_t *HostSigInfo, void *ucontext,
|
||||
GuestSigAction *GuestAction, stack_t *GuestStack,
|
||||
uint64_t NewGuestSP, const uint32_t eflags);
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
|
||||
enum class RestoreType {
|
||||
TYPE_REALTIME, ///< Signal restore type is from a `realtime` signal.
|
||||
TYPE_NONREALTIME, ///< Signal restore type is from a `non-realtime` signal.
|
||||
TYPE_PAUSE, ///< Signal restore type is from a GDB pause event.
|
||||
};
|
||||
|
||||
/*
|
||||
* Signal frames on 32-bit architecture needs to match exactly how the kernel generates the frame.
|
||||
* This is because large parts of the signal frame definition is part of the UAPI.
|
||||
* This means that when FEX sets up the signal frame, it needs to match the UAPI stack setup.
|
||||
*
|
||||
* The two signal stack frame types below describe the two different 32-bit frame types.
|
||||
*/
|
||||
|
||||
// The 32-bit non-realtime signal frame.
|
||||
// This frame type is used when the guest signal is used without the `SA_SIGINFO` flag.
|
||||
struct SigFrame_i32 {
|
||||
uint32_t pretcode; ///< sigreturn return branch point.
|
||||
int32_t Signal; ///< The signal hit.
|
||||
FEXCore::x86::sigcontext sc; ///< The signal context.
|
||||
x86::_libc_fpstate fpstate_unused; ///< Unused fpstate. Retained for backwards compatibility.
|
||||
uint32_t extramask[1]; ///< Upper 32-bits of the signal mask. Lower 32-bits is in the sigcontext.
|
||||
char retcode[8]; ///< Unused but needs to be filled. GDB seemingly uses as a debug marker.
|
||||
///< FP state now follows after this.
|
||||
};
|
||||
|
||||
// The 32-bit realtime signal frame.
|
||||
// This frame type is used when the guest signal is used with the `SA_SIGINFO` flag.
|
||||
struct RTSigFrame_i32 {
|
||||
uint32_t pretcode; ///< sigreturn return branch point.
|
||||
int32_t Signal; ///< The signal hit.
|
||||
uint32_t pinfo; ///< Pointer to siginfo_t
|
||||
uint32_t puc; ///< Pointer to ucontext_t
|
||||
FEXCore::x86::siginfo_t info;
|
||||
FEXCore::x86::ucontext_t uc;
|
||||
char retcode[8]; ///< Unused but needs to be filled. GDB seemingly uses as a debug marker.
|
||||
///< FP state now follows after this.
|
||||
};
|
||||
|
||||
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext, RestoreType Type);
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
|
||||
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
DispatcherConfig config;
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
@@ -11,14 +12,14 @@
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
|
||||
@@ -30,7 +31,7 @@ static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
: Dispatcher(ctx, config)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
|
||||
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
|
||||
FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true),
|
||||
nullptr) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
|
||||
@@ -169,36 +170,11 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
@@ -207,24 +183,15 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(rax);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rax, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(rax, qword [rax]);
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
L(AfterStore);
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
@@ -232,31 +199,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
@@ -264,27 +207,17 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(qword [rbx], rbx);
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
L(AfterStore);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else {
|
||||
jmp(rax);
|
||||
}
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -407,7 +340,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
End = Start + getSize();
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
fextl::string Name = fextl::fmt::format("Dispatch_{}", FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
@@ -416,15 +349,13 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
}
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
|
||||
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
|
||||
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
@@ -457,10 +388,11 @@ size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
Label InlineIRData;
|
||||
|
||||
|
||||
emit.mov(rdi, STATE);
|
||||
emit.lea(rsi, ptr[rip + InlineIRData]);
|
||||
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
@@ -475,7 +407,7 @@ size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
FEXCore::Allocator::VirtualFree(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -499,8 +431,8 @@ void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Threa
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<X86Dispatcher>(CTX, Config);
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return fextl::make_unique<X86Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,9 +1,24 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#define XBYAK64
|
||||
#define XBYAK_CUSTOM_ALLOC
|
||||
#define XBYAK_CUSTOM_MALLOC FEXCore::Allocator::malloc
|
||||
#define XBYAK_CUSTOM_FREE FEXCore::Allocator::free
|
||||
#define XBYAK_CUSTOM_SETS
|
||||
#define XBYAK_STD_UNORDERED_SET fextl::unordered_set
|
||||
#define XBYAK_STD_UNORDERED_MAP fextl::unordered_map
|
||||
#define XBYAK_STD_UNORDERED_MULTIMAP fextl::unordered_multimap
|
||||
#define XBYAK_STD_LIST fextl::list
|
||||
#define XBYAK_NO_EXCEPTION
|
||||
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
|
||||
+22
-17
@@ -7,6 +7,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
@@ -14,15 +15,13 @@ $end_info$
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
#include "Interface/Core/VSyscall/VSyscall.inc"
|
||||
@@ -304,18 +303,19 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
(Options.w && CTX->Config.Is64BitMode);
|
||||
const bool HasNarrowingDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
|
||||
|
||||
bool HasXMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasXMMDst = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_DST);
|
||||
bool HasMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasMMDst = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_DST);
|
||||
const bool HasXMMFlags = (Info->Flags & InstFlags::FLAGS_XMM_FLAGS) != 0;
|
||||
bool HasXMMSrc = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasXMMDst = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_DST);
|
||||
bool HasMMSrc = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasMMDst = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_DST);
|
||||
|
||||
// Is ModRM present via explicit instruction encoded or REX?
|
||||
const bool HasMODRM = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM);
|
||||
@@ -502,7 +502,12 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
|
||||
// If we have XMM flags at all, then SRC 1 cannot be a GPR. The only case where
|
||||
// this is possible is with BMI1 and BMI2 instructions (which are all GPR-based
|
||||
// and don't use XMM flags)
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMFlags);
|
||||
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
@@ -1108,7 +1113,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
|
||||
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
|
||||
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
fextl::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
+10
-9
@@ -1,14 +1,15 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <set>
|
||||
#include <stddef.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
@@ -29,7 +30,7 @@ public:
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
fextl::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
}
|
||||
|
||||
@@ -37,7 +38,7 @@ public:
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
void SetExternalBranches(fextl::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
@@ -89,10 +90,10 @@ private:
|
||||
uint64_t SymbolMinAddress {~0ULL};
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
|
||||
std::vector<DecodedBlocks> Blocks;
|
||||
std::set<uint64_t> BlocksToDecode;
|
||||
std::set<uint64_t> HasBlocks;
|
||||
std::set<uint64_t> *ExternalBranches {nullptr};
|
||||
fextl::vector<DecodedBlocks> Blocks;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t> *ExternalBranches {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
|
||||
+123
-117
@@ -8,11 +8,8 @@ $end_info$
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <iomanip>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
@@ -27,39 +24,45 @@ $end_info$
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
#ifndef _WIN32
|
||||
#include <elf.h>
|
||||
#include <netdb.h>
|
||||
#include <sys/socket.h>
|
||||
#endif
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <fstream>
|
||||
#include <fmt/format.h>
|
||||
#include <netdb.h>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <string_view>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "GdbServer.h"
|
||||
|
||||
namespace FEXCore
|
||||
{
|
||||
|
||||
#ifndef _WIN32
|
||||
void GdbServer::Break(int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (!CommsStream) {
|
||||
return;
|
||||
}
|
||||
|
||||
const auto str = fmt::format("S{:02x}", signal);
|
||||
const fextl::string str = fextl::fmt::format("S{:02x}", signal);
|
||||
SendPacket(*CommsStream, str);
|
||||
}
|
||||
|
||||
@@ -101,7 +104,7 @@ GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
StartThread();
|
||||
}
|
||||
|
||||
static int calculateChecksum(const std::string &packet) {
|
||||
static int calculateChecksum(const fextl::string &packet) {
|
||||
unsigned char checksum = 0;
|
||||
for (const char &c : packet) {
|
||||
checksum += c;
|
||||
@@ -109,8 +112,8 @@ static int calculateChecksum(const std::string &packet) {
|
||||
return checksum;
|
||||
}
|
||||
|
||||
static std::string hexstring(std::istringstream &ss, int delm) {
|
||||
std::string ret;
|
||||
static fextl::string hexstring(fextl::istringstream &ss, int delm) {
|
||||
fextl::string ret;
|
||||
|
||||
char hexString[3] = {0, 0, 0};
|
||||
while (ss.peek() != delm) {
|
||||
@@ -125,8 +128,8 @@ static std::string hexstring(std::istringstream &ss, int delm) {
|
||||
return ret;
|
||||
}
|
||||
|
||||
static std::string encodeHex(const unsigned char *data, size_t length) {
|
||||
std::ostringstream ss;
|
||||
static fextl::string encodeHex(const unsigned char *data, size_t length) {
|
||||
fextl::ostringstream ss;
|
||||
|
||||
for (size_t i=0; i < length; i++) {
|
||||
ss << std::setfill('0') << std::setw(2) << std::hex << int(data[i]);
|
||||
@@ -134,26 +137,19 @@ static std::string encodeHex(const unsigned char *data, size_t length) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static std::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
std::fstream fs(ThreadFile, std::fstream::in | std::fstream::binary);
|
||||
|
||||
if (fs.is_open()) {
|
||||
std::string ThreadName;
|
||||
fs >> ThreadName;
|
||||
fs.close();
|
||||
return ThreadName;
|
||||
}
|
||||
|
||||
return "<No Name>";
|
||||
static fextl::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
fextl::string ThreadName {"<No Name>"};
|
||||
FEXCore::FileLoading::LoadFile(ThreadName, ThreadFile);
|
||||
return ThreadName;
|
||||
}
|
||||
|
||||
// Packet parser
|
||||
// Takes a serial stream and reads a single packet
|
||||
// Un-escapes chars, checks the checksum and request a retransmit if it fails.
|
||||
// Once the checksum is validated, it acknowledges and returns the packet in a string
|
||||
std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
std::string packet{};
|
||||
fextl::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
fextl::string packet{};
|
||||
|
||||
// The GDB "Remote Serial Protocal" was originally 7bit clean for use on serial ports.
|
||||
// Binary data is useally hex encoded. However some later extentions just put
|
||||
@@ -172,7 +168,7 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
LogMan::Msg::EFmt("Dropping unexpected data: \"{}\"", packet);
|
||||
|
||||
// clear any existing data, must have been a mistake.
|
||||
packet = std::string();
|
||||
packet = fextl::string();
|
||||
break;
|
||||
case '}': // escape char
|
||||
{
|
||||
@@ -203,8 +199,8 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
return "";
|
||||
}
|
||||
|
||||
static std::string escapePacket(const std::string& packet) {
|
||||
std::ostringstream ss;
|
||||
static fextl::string escapePacket(const fextl::string& packet) {
|
||||
fextl::ostringstream ss;
|
||||
|
||||
for(const auto &c : packet) {
|
||||
switch (c) {
|
||||
@@ -225,9 +221,9 @@ static std::string escapePacket(const std::string& packet) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
void GdbServer::SendPacket(std::ostream &stream, const std::string& packet) {
|
||||
void GdbServer::SendPacket(std::ostream &stream, const fextl::string& packet) {
|
||||
const auto escaped = escapePacket(packet);
|
||||
const auto str = fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
const auto str = fextl::fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
|
||||
stream << str << std::flush;
|
||||
}
|
||||
@@ -263,7 +259,7 @@ struct FEX_PACKED GDBContextDefinition {
|
||||
uint32_t mxcsr;
|
||||
};
|
||||
|
||||
std::string GdbServer::readRegs() {
|
||||
fextl::string GdbServer::readRegs() {
|
||||
GDBContextDefinition GDB{};
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
@@ -311,11 +307,11 @@ std::string GdbServer::readRegs() {
|
||||
return encodeHex((unsigned char *)&GDB, sizeof(GDBContextDefinition));
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
|
||||
size_t addr;
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.get(); // Drop first letter
|
||||
ss >> std::hex >> addr;
|
||||
GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
size_t addr;
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.get(); // Drop first letter
|
||||
ss >> std::hex >> addr;
|
||||
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
@@ -395,8 +391,8 @@ GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
std::string buildTargetXML() {
|
||||
std::ostringstream xml;
|
||||
fextl::string buildTargetXML() {
|
||||
fextl::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
xml << "<!DOCTYPE target SYSTEM 'gdb-target.dtd'>\n";
|
||||
@@ -448,7 +444,7 @@ std::string buildTargetXML() {
|
||||
|
||||
// x87 stack
|
||||
for (int i=0; i < 8; i++) {
|
||||
reg("st" + std::to_string(i), "i387_ext", 80);
|
||||
reg(fextl::fmt::format("st{}", i), "i387_ext", 80);
|
||||
}
|
||||
|
||||
// x87 control
|
||||
@@ -484,7 +480,7 @@ std::string buildTargetXML() {
|
||||
|
||||
// SSE regs
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
reg("xmm" + std::to_string(i), "vec128", 128);
|
||||
reg(fextl::fmt::format("xmm{}", i), "vec128", 128);
|
||||
}
|
||||
|
||||
reg("mxcsr", "int", 32);
|
||||
@@ -520,8 +516,8 @@ std::string buildTargetXML() {
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
std::string buildOSData() {
|
||||
std::ostringstream xml;
|
||||
fextl::string buildOSData() {
|
||||
fextl::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
|
||||
@@ -541,24 +537,27 @@ void GdbServer::buildLibraryMap() {
|
||||
return;
|
||||
}
|
||||
|
||||
std::ostringstream xml;
|
||||
fextl::ostringstream xml;
|
||||
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
fextl::string MapsFile;
|
||||
FEXCore::FileLoading::LoadFile(MapsFile, "/proc/self/maps");
|
||||
fextl::istringstream MapsStream(MapsFile);
|
||||
|
||||
fextl::string Line;
|
||||
|
||||
struct FileData {
|
||||
uint64_t Begin;
|
||||
};
|
||||
|
||||
std::map<std::string, std::vector<FileData>> SegmentMaps;
|
||||
fextl::map<fextl::string, fextl::vector<FileData>> SegmentMaps;
|
||||
|
||||
// 7ff5dd6d2000-7ff5dd6d3000 rw-p 0000a000 103:0b 1881447 /usr/lib/x86_64-linux-gnu/libnss_compat.so.2
|
||||
std::string const &RuntimeExecutable = Filename();
|
||||
while (std::getline(fs, Line)) {
|
||||
auto ss = std::istringstream(Line);
|
||||
std::string Tmp;
|
||||
std::string Begin;
|
||||
std::string Name;
|
||||
fextl::string const &RuntimeExecutable = Filename();
|
||||
while (std::getline(MapsStream, Line)) {
|
||||
auto ss = fextl::istringstream(Line);
|
||||
fextl::string Tmp;
|
||||
fextl::string Begin;
|
||||
fextl::string Name;
|
||||
std::getline(ss, Begin, '-');
|
||||
std::getline(ss, Tmp, ' '); // End
|
||||
std::getline(ss, Tmp, ' '); // Perm
|
||||
@@ -609,18 +608,18 @@ void GdbServer::buildLibraryMap() {
|
||||
LibraryMapChanged = false;
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
std::string object;
|
||||
std::string rw;
|
||||
std::string annex;
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet) {
|
||||
fextl::string object;
|
||||
fextl::string rw;
|
||||
fextl::string annex;
|
||||
int annex_pid;
|
||||
int offset;
|
||||
int length;
|
||||
|
||||
// Parse Xfer message
|
||||
{
|
||||
auto ss = std::istringstream(packet);
|
||||
std::string expectXfer;
|
||||
auto ss = fextl::istringstream(packet);
|
||||
fextl::string expectXfer;
|
||||
char expectComma;
|
||||
|
||||
std::getline(ss, expectXfer, ':');
|
||||
@@ -631,7 +630,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
annex_pid = getpid();
|
||||
}
|
||||
else {
|
||||
auto ss_pid = std::istringstream(annex);
|
||||
auto ss_pid = fextl::istringstream(annex);
|
||||
ss_pid >> std::hex >> annex_pid;
|
||||
}
|
||||
ss >> std::hex >> offset;
|
||||
@@ -644,7 +643,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
}
|
||||
|
||||
// Lambda to correctly encode any reply
|
||||
auto encode = [&](std::string data) -> std::string {
|
||||
auto encode = [&](fextl::string data) -> fextl::string {
|
||||
if (offset == data.size())
|
||||
return "l";
|
||||
if (offset >= data.size())
|
||||
@@ -674,7 +673,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
auto Threads = CTX->GetThreads();
|
||||
|
||||
ThreadString.clear();
|
||||
std::ostringstream ss;
|
||||
fextl::ostringstream ss;
|
||||
ss << "<?xml version=\"1.0\"?>\n";
|
||||
ss << "<threads>\n";
|
||||
for (auto &Thread : *Threads) {
|
||||
@@ -710,7 +709,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
uint64_t auxv_ptr, auxv_size;
|
||||
CodeLoader->GetAuxv(auxv_ptr, auxv_size);
|
||||
std::string data;
|
||||
fextl::string data;
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
data.resize(auxv_size);
|
||||
memcpy(data.data(), reinterpret_cast<void*>(auxv_ptr), data.size());
|
||||
@@ -736,11 +735,14 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
|
||||
static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
uint64_t AddressEnd = Address + Size;
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
fextl::string MapsFile;
|
||||
FEXCore::FileLoading::LoadFile(MapsFile, "/proc/self/maps");
|
||||
fextl::istringstream MapsStream(MapsFile);
|
||||
|
||||
while (std::getline(fs, Line)) {
|
||||
if (fs.eof()) break;
|
||||
fextl::string Line;
|
||||
|
||||
while (std::getline(MapsStream, Line)) {
|
||||
if (MapsStream.eof()) break;
|
||||
uint64_t Begin, End;
|
||||
char r,w,x,p;
|
||||
if (sscanf(Line.c_str(), "%lx-%lx %c%c%c%c", &Begin, &End, &r, &w, &x, &p) == 6) {
|
||||
@@ -761,17 +763,17 @@ static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
GdbServer::HandledPacketType GdbServer::handleProgramOffsets() {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
uint64_t BaseOffset = CodeLoader->GetBaseOffset();
|
||||
auto str = fmt::format("Text={:x};Data={:x};Bss={:x}", BaseOffset, BaseOffset, BaseOffset);
|
||||
fextl::string str = fextl::fmt::format("Text={:x};Data={:x};Bss={:x}", BaseOffset, BaseOffset, BaseOffset);
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleMemory(const fextl::string &packet) {
|
||||
bool write;
|
||||
size_t addr;
|
||||
size_t length;
|
||||
std::string data;
|
||||
fextl::string data;
|
||||
|
||||
auto ss = std::istringstream(packet);
|
||||
auto ss = fextl::istringstream(packet);
|
||||
write = ss.get() == 'M';
|
||||
ss >> std::hex >> addr;
|
||||
ss.get(); // discard comma
|
||||
@@ -806,22 +808,22 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet)
|
||||
}
|
||||
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const std::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const fextl::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
|
||||
const auto split = [](const std::string &Str, char deliminator) -> std::vector<std::string> {
|
||||
std::vector<std::string> Elements;
|
||||
std::istringstream Input(Str);
|
||||
for (std::string line;
|
||||
const auto split = [](const fextl::string &Str, char deliminator) -> fextl::vector<fextl::string> {
|
||||
fextl::vector<fextl::string> Elements;
|
||||
fextl::istringstream Input(Str);
|
||||
for (fextl::string line;
|
||||
std::getline(Input, line);
|
||||
Elements.emplace_back(line));
|
||||
return Elements;
|
||||
};
|
||||
|
||||
if (match("QNonStop:")) {
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("QNonStop:").size());
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("QNonStop:").size());
|
||||
ss.get(); // discard colon
|
||||
ss >> NonStopMode;
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
@@ -832,7 +834,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
|
||||
// For feature documentation
|
||||
// https://sourceware.org/gdb/current/onlinedocs/gdb/General-Query-Packets.html#qSupported
|
||||
std::string SupportedFeatures{};
|
||||
fextl::string SupportedFeatures{};
|
||||
|
||||
// Required features
|
||||
SupportedFeatures += "PacketSize=32768;";
|
||||
@@ -901,7 +903,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
if (match("qfThreadInfo")) {
|
||||
auto Threads = CTX->GetThreads();
|
||||
|
||||
std::ostringstream ss;
|
||||
fextl::ostringstream ss;
|
||||
ss << "m";
|
||||
for (size_t i = 0; i < Threads->size(); ++i) {
|
||||
auto Thread = Threads->at(i);
|
||||
@@ -916,8 +918,8 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
return {"l", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qThreadExtraInfo")) {
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("qThreadExtraInfo").size());
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("qThreadExtraInfo").size());
|
||||
ss.get(); // discard comma
|
||||
uint32_t ThreadID;
|
||||
ss >> std::hex >> ThreadID;
|
||||
@@ -926,7 +928,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
}
|
||||
if (match("qC")) {
|
||||
// Returns the current Thread ID
|
||||
std::ostringstream ss;
|
||||
fextl::ostringstream ss;
|
||||
ss << "m" << std::hex << CTX->ParentThread->ThreadManager.TID;
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -935,10 +937,10 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qSymbol")) {
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("qSymbol").size());
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("qSymbol").size());
|
||||
ss.get(); // discard colon
|
||||
std::string Symbol_Val, Symbol_name;
|
||||
fextl::string Symbol_Val, Symbol_name;
|
||||
std::getline(ss, Symbol_Val, ':');
|
||||
std::getline(ss, Symbol_name, ':');
|
||||
|
||||
@@ -955,13 +957,13 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
std::fill(PassSignals.begin(), PassSignals.end(), false);
|
||||
|
||||
// eg: QPassSignals:e;10;14;17;1a;1b;1c;21;24;25;2c;4c;97;
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("QPassSignals").size());
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("QPassSignals").size());
|
||||
ss.get(); // discard colon
|
||||
|
||||
// We now have a semi-colon deliminated list of signals to pass to the guest process
|
||||
for (std::string tmp; std::getline(ss, tmp, ';'); ) {
|
||||
uint32_t Signal = std::stoi(tmp, nullptr, 16);
|
||||
for (fextl::string tmp; std::getline(ss, tmp, ';'); ) {
|
||||
uint32_t Signal = std::stoi(tmp.c_str(), nullptr, 16);
|
||||
if (Signal < SignalDelegator::MAX_SIGNALS) {
|
||||
PassSignals[Signal] = true;
|
||||
}
|
||||
@@ -984,7 +986,7 @@ GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid)
|
||||
case 's': {
|
||||
CTX->Step();
|
||||
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
|
||||
auto str = fmt::format("T05thread:{:02x};", getpid());
|
||||
fextl::string str = fextl::fmt::format("T05thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
@@ -1002,26 +1004,26 @@ GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid)
|
||||
}
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
const auto match = [&](const std::string& str) -> std::optional<std::istringstream> {
|
||||
GdbServer::HandledPacketType GdbServer::handleV(const fextl::string& packet) {
|
||||
const auto match = [&](const fextl::string& str) -> std::optional<fextl::istringstream> {
|
||||
if (packet.rfind(str, 0) == 0) {
|
||||
auto ss = std::istringstream(packet);
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(str.size());
|
||||
return ss;
|
||||
}
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
const auto F = [](int result) { return fmt::format("F{:x}", result); };
|
||||
const auto F_error = [] { return fmt::format("F-1,{:x}", errno); };
|
||||
const auto F_data = [](int result, const std::string& data) {
|
||||
const auto F = [](int result) -> fextl::string { return fextl::fmt::format("F{:x}", result); };
|
||||
const auto F_error = []() -> fextl::string { return fextl::fmt::format("F-1,{:x}", errno); };
|
||||
const auto F_data = [](int result, const fextl::string& data) -> fextl::string {
|
||||
// Binary encoded data is raw appended to the end
|
||||
return fmt::format("F{:#x};", result) + data;
|
||||
return fextl::fmt::format("F{:#x};", result) + data;
|
||||
};
|
||||
|
||||
std::optional<std::istringstream> ss;
|
||||
std::optional<fextl::istringstream> ss;
|
||||
if((ss = match("vFile:open:"))) {
|
||||
std::string filename;
|
||||
fextl::string filename;
|
||||
int flags;
|
||||
int mode;
|
||||
|
||||
@@ -1053,7 +1055,7 @@ GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
ss->get(); // discard comma
|
||||
*ss >> std::hex >> offset;
|
||||
|
||||
std::string data(count, '\0');
|
||||
fextl::string data(count, '\0');
|
||||
if (lseek(fd, offset, SEEK_SET) < 0) {
|
||||
return {F_error(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -1093,14 +1095,14 @@ GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
return {"", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleThreadOp(const fextl::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
|
||||
if (match("Hc")) {
|
||||
// Sets thread to this ID for stepping
|
||||
// This is deprecated and vCont should be used instead
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("Hc").size());
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("Hc").size());
|
||||
ss >> std::hex >> CurrentDebuggingThread;
|
||||
|
||||
CTX->Pause();
|
||||
@@ -1109,7 +1111,7 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet
|
||||
|
||||
if (match("Hg")) {
|
||||
// Sets thread for "other" operations
|
||||
auto ss = std::istringstream(packet);
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(std::string_view("Hg").size());
|
||||
ss >> std::hex >> CurrentDebuggingThread;
|
||||
|
||||
@@ -1121,8 +1123,8 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &packet) {
|
||||
auto ss = std::istringstream(packet);
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const fextl::string &packet) {
|
||||
auto ss = fextl::istringstream(packet);
|
||||
|
||||
// Don't do anything with set breakpoints yet
|
||||
[[maybe_unused]] bool Set{};
|
||||
@@ -1138,13 +1140,13 @@ GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &pack
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::ProcessPacket(const fextl::string &packet) {
|
||||
switch (packet[0]) {
|
||||
case '?': {
|
||||
// Indicates the reason that the thread has stopped
|
||||
// Behaviour changes if the target is in non-stop mode
|
||||
// Binja doesn't support S response here
|
||||
auto str = fmt::format("T00thread:{:x};", getpid());
|
||||
fextl::string str = fextl::fmt::format("T00thread:{:x};", getpid());
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 'c':
|
||||
@@ -1225,7 +1227,7 @@ void GdbServer::GdbServerLoop() {
|
||||
while ((c = CommsStream->get()) >= 0 ) {
|
||||
switch (c) {
|
||||
case '$': {
|
||||
std::string packet = ReadPacket(*CommsStream);
|
||||
auto packet = ReadPacket(*CommsStream);
|
||||
response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
@@ -1245,7 +1247,7 @@ void GdbServer::GdbServerLoop() {
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
CTX->Pause();
|
||||
auto str = fmt::format("T02thread:{:02x};", getpid());
|
||||
fextl::string str = fextl::fmt::format("T02thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
@@ -1279,7 +1281,8 @@ void GdbServer::StartThread() {
|
||||
}
|
||||
|
||||
void GdbServer::OpenListenSocket() {
|
||||
// open socket
|
||||
// getaddrinfo allocates memory that can't be removed.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
struct addrinfo hints, *res;
|
||||
|
||||
memset(&hints, 0, sizeof(hints));
|
||||
@@ -1308,9 +1311,11 @@ void GdbServer::OpenListenSocket() {
|
||||
}
|
||||
|
||||
listen(ListenSocket, 1);
|
||||
|
||||
freeaddrinfo(res);
|
||||
}
|
||||
|
||||
std::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
// Block until a connection arrives
|
||||
struct sockaddr_storage their_addr{};
|
||||
socklen_t addr_size{};
|
||||
@@ -1318,7 +1323,8 @@ std::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
return std::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
return fextl::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore
|
||||
+20
-19
@@ -8,13 +8,14 @@ $end_info$
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -37,10 +38,10 @@ private:
|
||||
void Break(int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
std::unique_ptr<std::iostream> OpenSocket();
|
||||
fextl::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
std::string ReadPacket(std::iostream &stream);
|
||||
void SendPacket(std::ostream &stream, const std::string& packet);
|
||||
fextl::string ReadPacket(std::iostream &stream);
|
||||
void SendPacket(std::ostream &stream, const fextl::string& packet);
|
||||
|
||||
void SendACK(std::ostream &stream, bool NACK);
|
||||
|
||||
@@ -48,7 +49,7 @@ private:
|
||||
void WaitForThreadWakeup();
|
||||
|
||||
struct HandledPacketType {
|
||||
std::string Response{};
|
||||
fextl::string Response{};
|
||||
enum ResponseType {
|
||||
TYPE_NONE,
|
||||
TYPE_UNKNOWN,
|
||||
@@ -61,32 +62,32 @@ private:
|
||||
};
|
||||
|
||||
void SendPacketPair(const HandledPacketType& packetPair);
|
||||
HandledPacketType ProcessPacket(const std::string &packet);
|
||||
HandledPacketType handleQuery(const std::string &packet);
|
||||
HandledPacketType handleXfer(const std::string &packet);
|
||||
HandledPacketType handleMemory(const std::string &packet);
|
||||
HandledPacketType handleV(const std::string& packet);
|
||||
HandledPacketType handleThreadOp(const std::string &packet);
|
||||
HandledPacketType handleBreakpoint(const std::string &packet);
|
||||
HandledPacketType ProcessPacket(const fextl::string &packet);
|
||||
HandledPacketType handleQuery(const fextl::string &packet);
|
||||
HandledPacketType handleXfer(const fextl::string &packet);
|
||||
HandledPacketType handleMemory(const fextl::string &packet);
|
||||
HandledPacketType handleV(const fextl::string& packet);
|
||||
HandledPacketType handleThreadOp(const fextl::string &packet);
|
||||
HandledPacketType handleBreakpoint(const fextl::string &packet);
|
||||
HandledPacketType handleProgramOffsets();
|
||||
|
||||
HandledPacketType ThreadAction(char action, uint32_t tid);
|
||||
|
||||
std::string readRegs();
|
||||
HandledPacketType readReg(const std::string& packet);
|
||||
fextl::string readRegs();
|
||||
HandledPacketType readReg(const fextl::string& packet);
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
std::unique_ptr<std::iostream> CommsStream;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
fextl::unique_ptr<std::iostream> CommsStream;
|
||||
std::mutex sendMutex;
|
||||
bool SettingNoAckMode{false};
|
||||
bool NoAckMode{false};
|
||||
bool NonStopMode{false};
|
||||
std::string ThreadString{};
|
||||
std::string OSDataString{};
|
||||
fextl::string ThreadString{};
|
||||
fextl::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::string LibraryMapString{};
|
||||
fextl::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
|
||||
+21
-9
@@ -9,7 +9,7 @@
|
||||
#endif
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <xbyak/xbyak_util.h>
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -17,12 +17,12 @@ namespace FEXCore {
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
// Log2 of the blocksize in 32-bit words
|
||||
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static uint32_t GetDCZID() {
|
||||
[[maybe_unused]] static uint32_t GetDCZID() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], DCZID_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
@@ -54,7 +54,12 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
#ifndef _WIN32
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
// Need to use ID registers in WINE.
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
@@ -133,13 +138,14 @@ HostFeatures::HostFeatures() {
|
||||
SupportsPMULL_128Bit = Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0000, data);
|
||||
if (data[0] >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0008, data);
|
||||
SupportsCLZERO = data[1] & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
@@ -160,5 +166,11 @@ HostFeatures::HostFeatures() {
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Disable AVX if the configuration explicitly has disabled it.
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
if (!EnableAVX) {
|
||||
SupportsAVX = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -143,6 +143,15 @@ DEF_OP(CPUID) {
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
uint32_t *DstPtr = GetDest<uint32_t*>(Data->SSAData, Node);
|
||||
const uint32_t Function = *GetSrc<uint32_t*>(Data->SSAData, Op->Function);
|
||||
|
||||
auto Results = Data->State->CTX->RunXCRFunction(Function);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,7 +8,7 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include "F80Ops.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/F80Fallbacks.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
@@ -417,7 +417,6 @@ DEF_OP(F64SCALE) {
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+3
-7
@@ -4,12 +4,9 @@
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
template<IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
static X80SoftFloat handle4(float src) {
|
||||
@@ -395,5 +392,4 @@ struct OpHandlers<IR::OP_F80LOADFCW> {
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
+75
@@ -0,0 +1,75 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
enum IROps : uint8_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// Base template for fallback handling.
|
||||
//
|
||||
// Registering and hooking up fallback is currently like so:
|
||||
//
|
||||
// 1. Go to InterpreterFallbacks.cpp and create a template specialization of
|
||||
// the GetFallbackInfo member function.
|
||||
//
|
||||
// This member function should reasonably define what the fallback you're
|
||||
// going to create will take as parameters and return as a result. For example:
|
||||
//
|
||||
// template<>
|
||||
// FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), Core::FallbackHandlerIndex Index) {
|
||||
// return {FABI_F80_F64, (void*)fn, Index};
|
||||
// }
|
||||
//
|
||||
// Defines info about a fallback that takes a double as an argument and
|
||||
// returns a X80SoftFloat instance.
|
||||
//
|
||||
// You will also want to define a new FallbackHandlerIndex enum member and use it
|
||||
// to set up the new info handler into the Info array in FillFallbackIndexPointers.
|
||||
//
|
||||
// 1.1. (potentially optional). Define a new ABI element in the FallbackAPI enum.
|
||||
// This ABI enum value will be used to tell the JITs how to handle the fallback
|
||||
// properly. These enum values specify the return type followed by its argument types.
|
||||
//
|
||||
// So, FABI_I64_F80_F80, for example indicates that the function will behave like a
|
||||
// function as if were defined as:
|
||||
//
|
||||
// uint64_t fn(X80SoftFloat, X80SoftFloat)
|
||||
//
|
||||
// 1.2. (potentially optional). If you needed to define a new enum ABI type like in 1.1, then
|
||||
// you need to add the handling for it in the JITs, which can be found in the respective
|
||||
// JIT's JIT.cpp file in a function called Op_Unhandled
|
||||
//
|
||||
// You need to add a new case to the ABI switch statement using the new ABI type
|
||||
// and do the necessary moving of data from register-allocated JIT parameters
|
||||
// into that platform's registers that respects the calling convention. After this is
|
||||
// done, most of the necessary background boilerplate is finished.
|
||||
//
|
||||
// 2. Now, make a specialization of this class with a member function named 'handle()'
|
||||
// that takes the same parameters as the ones described in the fallback info function
|
||||
// specialization.
|
||||
//
|
||||
// For example, if you have the fallback info from the example in step 1, it would be:
|
||||
//
|
||||
// template <>
|
||||
// struct OpHandlers<IR::CoolNewIROpcode> {
|
||||
// static X80SoftFloat handle(double src) {
|
||||
// return ...;
|
||||
// }
|
||||
// };
|
||||
//
|
||||
// 3. Fill out the behavior of the OpHandler specialization to perform what you would like
|
||||
// the fallback to do.
|
||||
//
|
||||
// 4. Add an implementation of the IR op to the Interpreter that passes through to the
|
||||
// OpHandler implementation.
|
||||
//
|
||||
// 5. Done.
|
||||
//
|
||||
template <IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+25
-2
@@ -1,6 +1,8 @@
|
||||
#include "FEXCore/Core/CoreState.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/F80Ops.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/F80Fallbacks.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -87,6 +89,16 @@ FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXC
|
||||
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint32_t(*fn)(uint64_t, uint64_t, __uint128_t, __uint128_t, uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_I64_I64_I128_I128_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint32_t(*fn)(__uint128_t, __uint128_t, uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_I128_I128_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
|
||||
@@ -144,6 +156,9 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
|
||||
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
|
||||
|
||||
// SSE4.2 string instructions
|
||||
Info[Core::OPINDEX_VPCMPESTRX] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX).fn);
|
||||
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX).fn);
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
@@ -302,6 +317,14 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
// SSE4.2 Fallbacks
|
||||
case IR::OP_VPCMPESTRX:
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX);
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX);
|
||||
return true;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
+408
@@ -0,0 +1,408 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
enum class SourceData {
|
||||
U8,
|
||||
U16,
|
||||
S8,
|
||||
S16,
|
||||
};
|
||||
|
||||
enum class Polarity {
|
||||
Positive,
|
||||
Negative,
|
||||
PositiveMasked,
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetExplicitLength(RAX, control) - 1;
|
||||
const auto valid_rhs = GetExplicitLength(RDX, control) - 1;
|
||||
|
||||
return MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const uint32_t upper_limit = (16U >> (control & 1)) - 1;
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
// [OF | CF | SF | ZF]
|
||||
uint32_t flags = 0;
|
||||
flags |= (valid_rhs < upper_limit) ? 0b01 : 0b00;
|
||||
flags |= (valid_lhs < upper_limit) ? 0b10 : 0b00;
|
||||
|
||||
const uint32_t result = HandlePolarity(aggregation, control, upper_limit, valid_rhs);
|
||||
if (result != 0) {
|
||||
flags |= 0b0100;
|
||||
}
|
||||
if ((result & 1) != 0) {
|
||||
flags |= 0b1000;
|
||||
}
|
||||
|
||||
// We tack the flags on top of the result to avoid needing to handle
|
||||
// multiple return values in the JITs.
|
||||
return result | (flags << 16);
|
||||
}
|
||||
|
||||
static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
|
||||
// Bit 8 controls whether or not the reg value is 64-bit or 32-bit.
|
||||
int64_t value = 0;
|
||||
if (((control >> 8) & 1) != 0) {
|
||||
value = static_cast<int64_t>(reg);
|
||||
} else {
|
||||
// We need a sign extend in this case.
|
||||
value = static_cast<int32_t>(reg);
|
||||
}
|
||||
|
||||
// If control[0] is set, then we're dealing with words instead of bytes
|
||||
const int64_t limit = (control & 1) != 0 ? 8 : 16;
|
||||
|
||||
// Length needs to saturate to 16 (if bytes) or 8 (if words)
|
||||
// when the length value is greater than 16 (if bytes)/8 (if words)
|
||||
// or if the length value is less than -16 (if bytes)/-8 (if words).
|
||||
if (value < -limit || value > limit) {
|
||||
return limit;
|
||||
}
|
||||
|
||||
return std::abs(static_cast<int>(value));
|
||||
}
|
||||
|
||||
static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
|
||||
const auto* vec_ptr = reinterpret_cast<const uint8_t*>(&vec);
|
||||
|
||||
// Control bits [1:0] define the data type being dealt with.
|
||||
switch (static_cast<SourceData>(control & 0b11)) {
|
||||
case SourceData::U8:
|
||||
return static_cast<int32_t>(vec_ptr[index]);
|
||||
case SourceData::U16: {
|
||||
uint16_t value{};
|
||||
std::memcpy(&value, vec_ptr + (sizeof(uint16_t) * static_cast<size_t>(index)), sizeof(value));
|
||||
return value;
|
||||
}
|
||||
case SourceData::S8:
|
||||
return static_cast<int8_t>(vec_ptr[index]);
|
||||
case SourceData::S16:
|
||||
default: {
|
||||
int16_t value{};
|
||||
std::memcpy(&value, vec_ptr + (sizeof(int16_t) * static_cast<size_t>(index)), sizeof(value));
|
||||
return value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint32_t PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
switch (static_cast<AggregationOp>((control >> 2) & 0b11)) {
|
||||
case AggregationOp::EqualAny:
|
||||
return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::Ranges:
|
||||
return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualEach:
|
||||
return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualOrdered:
|
||||
default:
|
||||
return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
}
|
||||
|
||||
static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
|
||||
switch (static_cast<Polarity>((control >> 4) & 0b11)) {
|
||||
case Polarity::Negative:
|
||||
return value ^ ((2U << upper_limit) - 1);
|
||||
case Polarity::NegativeMasked:
|
||||
return value ^ ((1U << (valid_rhs + 1)) - 1);
|
||||
case Polarity::Positive:
|
||||
case Polarity::PositiveMasked:
|
||||
default:
|
||||
// Both positive masking and positive polarity are documented
|
||||
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
|
||||
// is our 'value' parameter, so we don't need to do anything in
|
||||
// these cases except return the same value.
|
||||
return value;
|
||||
}
|
||||
}
|
||||
|
||||
// Finds characters from an overall character set.
|
||||
//
|
||||
// Scans through RHS trying to find any characters contained in LHS.
|
||||
// Think of this as a sort of vectorized version of strspn (kind of).
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [a, b, c, d, e, f, g, n]
|
||||
// RHS -> [z, k, v, c, d, o, p, n]
|
||||
//
|
||||
// With both explicit lengths for each string being 8 (the max length for words),
|
||||
// this would result in an intermediate result like:
|
||||
//
|
||||
// 0b1001'1000
|
||||
// │ │ │
|
||||
// 'n' match ───┘ │ │
|
||||
// │ │
|
||||
// 'd' match ──────┘ │
|
||||
// │
|
||||
// 'c' match ────────┘
|
||||
//
|
||||
static uint32_t HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
uint32_t result = 0;
|
||||
|
||||
for (int j = valid_rhs; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
const int rhs_value = GetElement(rhs, j, control);
|
||||
for (int i = valid_lhs; i >= 0; i--) {
|
||||
const int lhs_value = GetElement(lhs, i, control);
|
||||
result |= static_cast<uint32_t>(rhs_value == lhs_value);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Determines if a character falls within a limited range
|
||||
//
|
||||
// Scans through rhs using a range denoted by two elements
|
||||
// in lhs and determines if the respective character in rhs
|
||||
// falls within its range.
|
||||
//
|
||||
// i.e.
|
||||
// lhs_upper_bound >= rhs_value && lhs_lower_bound <= rhs_value
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [a, z, A, Z, 0, 0, 0, 0]
|
||||
// RHS -> [z, k, ., C, M, ;, \, ']
|
||||
//
|
||||
// With LHS's length being 4 and RHS's lenth being 8,
|
||||
// this would result in an intermediate result like:
|
||||
//
|
||||
// 0b0001'1011
|
||||
// │ │ ││
|
||||
// 'z' >= 'M' && 'a' <= 'M' ─────┘ │ ││
|
||||
// │ ││
|
||||
// 'z' >= 'C' && 'a' <= 'C' ───────┘ ││
|
||||
// ││
|
||||
// 'Z' >= 'k' && 'A' <= 'k' ─────────┘│
|
||||
// │
|
||||
// 'Z' >= 'z' && 'A' <= 'z' ──────────┘
|
||||
//
|
||||
static uint32_t HandleRanges(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
uint32_t result = 0;
|
||||
|
||||
for (int j = valid_rhs; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
const int element = GetElement(rhs, j, control);
|
||||
for (int i = (valid_lhs - 1) | 1; i >= 0; i -= 2) {
|
||||
const int upper_bound = GetElement(lhs, i - 0, control);
|
||||
const int lower_bound = GetElement(lhs, i - 1, control);
|
||||
|
||||
const bool ge = upper_bound >= element;
|
||||
const bool le = lower_bound <= element;
|
||||
|
||||
result |= static_cast<uint32_t>(ge && le);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Determines if each character is equal to one another (string compare)
|
||||
//
|
||||
// Essentially the PCMPXSTRX variant of memcmp/strcmp. Sets the bit of the
|
||||
// resulting mask if both elements are equal to one another. Otherwise
|
||||
// sets it to false.
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [a, b, c, d, e, f, g, n]
|
||||
// RHS -> [a, b, c, d, e, f, e, x]
|
||||
//
|
||||
// With both explicit lengths for each string being 8 (the max length for words),
|
||||
// this would result in an intermediate result like:
|
||||
//
|
||||
// 0b0011'1111
|
||||
// ││ ││││
|
||||
// 'f' == 'f' ────┘│ ││││
|
||||
// │ ││││
|
||||
// 'e' == 'e' ─────┘ ││││
|
||||
// ││││
|
||||
// 'd' == 'd' ───────┘│││
|
||||
// │││
|
||||
// 'c' == 'c' ────────┘││
|
||||
// ││
|
||||
// 'b' == 'b' ─────────┘│
|
||||
// │
|
||||
// 'a' == 'a' ──────────┘
|
||||
//
|
||||
static uint32_t HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
const auto upper_limit = (16 >> (control & 1)) - 1;
|
||||
const auto max_valid = std::max(valid_lhs, valid_rhs);
|
||||
const auto min_valid = std::min(valid_lhs, valid_rhs);
|
||||
|
||||
// All values past the end of string must be forced to true.
|
||||
// (See 4.1.6 Valid/Invalid Override of Comparisons in the Intel Software Development Manual)
|
||||
// So we can calculate this part of the mask ahead of time and set all those to-be bits to true
|
||||
// and then progressively shift them into place over the course of execution.
|
||||
uint32_t result = (1U << (upper_limit - max_valid)) - 1;
|
||||
result <<= (max_valid - min_valid);
|
||||
|
||||
for (int i = min_valid; i >= 0; i--) {
|
||||
const int lhs_element = GetElement(lhs, i, control);
|
||||
const int rhs_element = GetElement(rhs, i, control);
|
||||
|
||||
result <<= 1;
|
||||
result |= static_cast<uint32_t>(lhs_element == rhs_element);
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Determines if a substring exists within an overall string
|
||||
//
|
||||
// Somewhat equivalent to the behavior of strstr.
|
||||
//
|
||||
// Sets the corresponding index in the result where a substring is found.
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [b, a, x, z, y, v, o, m]
|
||||
// RHS -> [b, a, d, b, a, n, k, s]
|
||||
//
|
||||
// With the length of LHS being 2 and the length of RHS being 8, we have a composition like:
|
||||
//
|
||||
// Substring to look for
|
||||
// ┌──┴──┐
|
||||
// LHS -> [b, a, x, z, y, v, o, m]
|
||||
// RHS -> [b, a, d, b, a, n, k, s]
|
||||
// └───────────┬────────────┘
|
||||
// Entire string to search
|
||||
//
|
||||
// And we end up with a result like:
|
||||
//
|
||||
// 0b0000'1001
|
||||
// │ │
|
||||
// At index 3 ───────┘ │
|
||||
// │
|
||||
// At index 0 ──────────┘
|
||||
//
|
||||
static uint32_t HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
const auto upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Edge case!
|
||||
// If we have *no* valid characters in our inner string, then
|
||||
// we need to return the intermediate result as
|
||||
// 0xFF (if operating on words) or 0xFFFF (if operating on bytes)
|
||||
if (valid_lhs == -1) {
|
||||
return (2U << upper_limit) - 1;
|
||||
}
|
||||
|
||||
uint32_t result = 0;
|
||||
const int initial = valid_rhs == upper_limit ? valid_rhs
|
||||
: valid_rhs - valid_lhs;
|
||||
for (int j = initial; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
uint32_t value = 1;
|
||||
const int start = std::min(valid_rhs - j, valid_lhs);
|
||||
for (int i = start; i >= 0; i--) {
|
||||
const int lhs_value = GetElement(lhs, i + 0, control);
|
||||
const int rhs_value = GetElement(rhs, i + j, control);
|
||||
|
||||
value &= static_cast<uint32_t>(lhs_value == rhs_value);
|
||||
}
|
||||
|
||||
result |= value;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element{};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -6,25 +6,22 @@
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Dispatcher;
|
||||
class X86DispatchGenerator;
|
||||
class Arm64DispatchGenerator;
|
||||
|
||||
#define DESTMAP_AS_MAP 0
|
||||
#if DESTMAP_AS_MAP
|
||||
using DestMapType = std::unordered_map<uint32_t, uint32_t>;
|
||||
#else
|
||||
using DestMapType = std::vector<uint32_t>;
|
||||
#endif
|
||||
using DestMapType = fextl::vector<uint32_t>;
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(Dispatcher *Dispatch,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
[[nodiscard]] fextl::string GetName() override { return "Interpreter"; }
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
@@ -36,7 +33,7 @@ public:
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
private:
|
||||
|
||||
@@ -1,17 +1,14 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <memory>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
@@ -49,18 +46,6 @@ InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::Internal
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return reinterpret_cast<Context::ContextImpl*>(Thread->CTX)->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
|
||||
@@ -103,12 +88,8 @@ void InterpreterCore::ClearCache() {
|
||||
BufferUsed = 0;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::ContextImpl *CTX) {
|
||||
InterpreterCore::InitializeSignalHandlers(CTX);
|
||||
fextl::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return fextl::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
@@ -14,7 +15,7 @@ namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
struct DispatcherConfig;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx,
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures();
|
||||
|
||||
@@ -2,11 +2,6 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "InterpreterDefines.h"
|
||||
#include "InterpreterOps.h"
|
||||
#include "F80Ops.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -123,6 +118,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
@@ -154,7 +150,10 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
@@ -269,6 +268,8 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
|
||||
REGISTER_OP(VPCMPISTRX, VPCMPISTRX);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
|
||||
@@ -36,6 +36,8 @@ namespace FEXCore::CPU {
|
||||
FABI_I64_F80_F80,
|
||||
FABI_F80_F80,
|
||||
FABI_F80_F80_F80,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
@@ -152,6 +154,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -181,7 +184,10 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
@@ -288,6 +294,8 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VPCMPESTRX);
|
||||
DEF_OP(VPCMPISTRX);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
|
||||
+271
-16
@@ -288,11 +288,134 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto NumElements = OpSize / ElementSize;
|
||||
|
||||
const auto *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
const auto *Mask = GetSrc<uint8_t const*>(Data->SSAData, Op->Mask);
|
||||
|
||||
const auto SetElements = [NumElements]<typename T>(void* Dst, const T* MaskValues, const T* MemoryData) {
|
||||
const auto SignBit = 1ULL << ((sizeof(T) * 8) - 1);
|
||||
for (size_t i = 0; i < NumElements; i++) {
|
||||
if ((MaskValues[i] & SignBit) != 0) {
|
||||
std::memcpy(static_cast<uint8_t*>(Dst) + (i * sizeof(T)), MemoryData + i, sizeof(T));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
|
||||
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
SetElements(GDP, Mask, MemData);
|
||||
return;
|
||||
}
|
||||
case 2: {
|
||||
SetElements(GDP,
|
||||
reinterpret_cast<const uint16_t*>(Mask),
|
||||
reinterpret_cast<const uint16_t*>(MemData));
|
||||
return;
|
||||
}
|
||||
case 4: {
|
||||
SetElements(GDP,
|
||||
reinterpret_cast<const uint32_t*>(Mask),
|
||||
reinterpret_cast<const uint32_t*>(MemData));
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
SetElements(GDP,
|
||||
reinterpret_cast<const uint64_t*>(Mask),
|
||||
reinterpret_cast<const uint64_t*>(MemData));
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VLoadVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VStoreVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto NumElements = OpSize / ElementSize;
|
||||
|
||||
auto *Dst = *GetSrc<uint8_t**>(Data->SSAData, Op->Addr);
|
||||
const auto *RegData = GetSrc<uint8_t const*>(Data->SSAData, Op->Data);
|
||||
const auto *Mask = GetSrc<uint8_t const*>(Data->SSAData, Op->Mask);
|
||||
|
||||
const auto SetElements = [NumElements]<typename T>(void* Dst, const T* MaskValues, const T* DataVals) {
|
||||
const auto SignBit = 1ULL << ((sizeof(T) * 8) - 1);
|
||||
for (size_t i = 0; i < NumElements; i++) {
|
||||
if ((MaskValues[i] & SignBit) != 0) {
|
||||
std::memcpy(static_cast<uint8_t*>(Dst) + (i * sizeof(T)), DataVals + i, sizeof(T));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: Dst += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: Dst += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: Dst += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
SetElements(Dst, Mask, RegData);
|
||||
return;
|
||||
}
|
||||
case 2: {
|
||||
SetElements(Dst,
|
||||
reinterpret_cast<const uint16_t*>(Mask),
|
||||
reinterpret_cast<const uint16_t*>(RegData));
|
||||
return;
|
||||
}
|
||||
case 4: {
|
||||
SetElements(Dst,
|
||||
reinterpret_cast<const uint32_t*>(Mask),
|
||||
reinterpret_cast<const uint32_t*>(RegData));
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
SetElements(Dst,
|
||||
reinterpret_cast<const uint64_t*>(Mask),
|
||||
reinterpret_cast<const uint64_t*>(RegData));
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VStoreVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemSet) {
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
const int32_t Size = Op->Size;
|
||||
|
||||
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
|
||||
uint64_t MemPrefix{};
|
||||
if (!Op->Prefix.IsInvalid()) {
|
||||
MemPrefix = *GetSrc<uint64_t*>(Data->SSAData, Op->Prefix);
|
||||
}
|
||||
|
||||
const auto Value = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
const auto Length = *GetSrc<uint64_t*>(Data->SSAData, Op->Length);
|
||||
const auto Direction = *GetSrc<uint8_t*>(Data->SSAData, Op->Direction);
|
||||
@@ -313,16 +436,16 @@ DEF_OP(MemSet) {
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint8_t>*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint8_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint16_t>*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint16_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint32_t>*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint32_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint64_t>*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint64_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
@@ -332,16 +455,16 @@ DEF_OP(MemSet) {
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElements(reinterpret_cast<uint8_t*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<uint8_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElements(reinterpret_cast<uint16_t*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<uint16_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElements(reinterpret_cast<uint32_t*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<uint32_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElements(reinterpret_cast<uint64_t*>(MemData), Value, Length);
|
||||
MemSetElements(reinterpret_cast<uint64_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
@@ -354,16 +477,16 @@ DEF_OP(MemSet) {
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint8_t>*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint8_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint16_t>*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint16_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint32_t>*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint32_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint64_t>*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint64_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
@@ -373,16 +496,16 @@ DEF_OP(MemSet) {
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsInverse(reinterpret_cast<uint8_t*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<uint8_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsInverse(reinterpret_cast<uint16_t*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<uint16_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsInverse(reinterpret_cast<uint32_t*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<uint32_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsInverse(reinterpret_cast<uint64_t*>(MemData), Value, Length);
|
||||
MemSetElementsInverse(reinterpret_cast<uint64_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
@@ -393,6 +516,138 @@ DEF_OP(MemSet) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
const int32_t Size = Op->Size;
|
||||
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
|
||||
char *MemDataDest = *GetSrc<char **>(Data->SSAData, Op->AddrDest);
|
||||
char *MemDataSrc = *GetSrc<char **>(Data->SSAData, Op->AddrSrc);
|
||||
|
||||
uint64_t DestPrefix{};
|
||||
uint64_t SrcPrefix{};
|
||||
if (!Op->PrefixDest.IsInvalid()) {
|
||||
DestPrefix = *GetSrc<uint64_t*>(Data->SSAData, Op->PrefixDest);
|
||||
|
||||
}
|
||||
if (!Op->PrefixSrc.IsInvalid()) {
|
||||
SrcPrefix = *GetSrc<uint64_t*>(Data->SSAData, Op->PrefixSrc);
|
||||
}
|
||||
|
||||
const auto Length = *GetSrc<uint64_t*>(Data->SSAData, Op->Length);
|
||||
const auto Direction = *GetSrc<uint8_t*>(Data->SSAData, Op->Direction);
|
||||
|
||||
auto MemSetElementsAtomic = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[i].store(MemSrc[i].load());
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElementsAtomicInverse = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[-i].store(MemSrc[-i].load());
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElements = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[i] = MemSrc[i];
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElementsInverse = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[-i] = MemSrc[-i];
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 0) { // Forward
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint8_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint8_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint16_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint16_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint32_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint32_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint64_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint64_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElements(reinterpret_cast<uint8_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint8_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElements(reinterpret_cast<uint16_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint16_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElements(reinterpret_cast<uint32_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint32_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElements(reinterpret_cast<uint64_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint64_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
DstPtr[0] = reinterpret_cast<uint64_t>(MemDataDest + (Length * Size));
|
||||
DstPtr[1] = reinterpret_cast<uint64_t>(MemDataSrc + (Length * Size));
|
||||
}
|
||||
else { // Backward
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint8_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint8_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint16_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint16_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint32_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint32_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint64_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint64_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsInverse(reinterpret_cast<uint8_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint8_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsInverse(reinterpret_cast<uint16_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint16_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsInverse(reinterpret_cast<uint32_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint32_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsInverse(reinterpret_cast<uint64_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint64_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
DstPtr[0] = reinterpret_cast<uint64_t>(MemDataDest - (Length * Size));
|
||||
DstPtr[1] = reinterpret_cast<uint64_t>(MemDataSrc - (Length * Size));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
|
||||
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
|
||||
@@ -2276,6 +2278,37 @@ DEF_OP(VRev64) {
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VPCMPESTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
|
||||
const auto RAX = *GetSrc<uint64_t*>(Data->SSAData, Op->RAX);
|
||||
const auto RDX = *GetSrc<uint64_t*>(Data->SSAData, Op->RDX);
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPESTRX>::handle(RAX, RDX, LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
memcpy(GDP, &Result, sizeof(Result));
|
||||
}
|
||||
|
||||
DEF_OP(VPCMPISTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPISTRX>::handle(LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
memcpy(GDP, &Result, sizeof(Result));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -494,7 +494,7 @@ DEF_OP(PDep) {
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, SpillCode);
|
||||
SpillStaticRegs(TMP1, false, SpillCode);
|
||||
|
||||
|
||||
mov(EmitSize, InputReg, Input);
|
||||
@@ -558,7 +558,7 @@ DEF_OP(PExt) {
|
||||
// We sadly need to spill a reg for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, 1U << Mask.Idx());
|
||||
SpillStaticRegs(TMP2, false, 1U << Mask.Idx());
|
||||
mov(EmitSize, Mask, ZeroReg);
|
||||
|
||||
// Main loop
|
||||
|
||||
@@ -22,7 +22,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
@@ -175,7 +175,7 @@ DEF_OP(Syscall) {
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(true, GPRSpillMask, FPRSpillMask);
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -216,8 +216,11 @@ DEF_OP(Syscall) {
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Move result to its destination register
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -257,7 +260,7 @@ DEF_OP(InlineSyscall) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -325,7 +328,7 @@ DEF_OP(Thunk) {
|
||||
// X0: CTX
|
||||
// X1: Args (from guest stack)
|
||||
|
||||
SpillStaticRegs(); // spill to ctx before ra64 spill
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -400,12 +403,12 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X1: RIP
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
@@ -421,7 +424,7 @@ DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
@@ -447,6 +450,34 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
+126
-68
@@ -14,8 +14,6 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
@@ -25,7 +23,6 @@ $end_info$
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
@@ -33,7 +30,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
#include <stdio.h>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
@@ -87,7 +83,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -107,7 +103,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F32:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
@@ -131,7 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -157,7 +153,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -187,7 +183,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -213,7 +209,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -239,7 +235,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -263,7 +259,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -289,7 +285,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -314,7 +310,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -339,7 +335,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -364,7 +360,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -392,7 +388,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -419,7 +415,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -449,6 +445,78 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r4, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r5, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r6, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x7, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r7);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r7);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), ARMEmitter::WReg::w0);
|
||||
break;
|
||||
}
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), ARMEmitter::WReg::w0);
|
||||
break;
|
||||
}
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -517,20 +585,16 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
uint32_t NumUsedGPRs = NumGPRs;
|
||||
uint32_t NumUsedGPRPairs = NumGPRPairs;
|
||||
uint32_t UsedRegisterCount = RegisterCount;
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
|
||||
RAPass->AllocateRegisterSet(UsedRegisterCount, RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, NumUsedGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, SRA64.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, NumFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, SRAFPR.size() );
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, NumUsedGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, ConfiguredGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, ConfiguredSRAGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, ConfiguredFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, ConfiguredSRAFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, ConfiguredGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < NumUsedGPRPairs; ++i) {
|
||||
for (uint32_t i = 0; i < ConfiguredGPRPairs; ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
@@ -551,6 +615,11 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
@@ -570,23 +639,25 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return reinterpret_cast<Context::ContextImpl*>(Thread->CTX)->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
// Setup dynamic dispatch.
|
||||
if (CTX->Dispatcher->GetConfig().StaticRegisterAllocation) {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegisterSRA;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegisterSRA;
|
||||
}
|
||||
else {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegister;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegister;
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(static_cast<Context::ContextImpl*>(Thread->CTX)->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
}
|
||||
else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -767,6 +838,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
@@ -844,6 +916,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
@@ -873,8 +946,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
@@ -883,24 +956,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
case FEXCore::IR::IROps::OP_LOADMEMTSO:
|
||||
if (ParanoidTSO()) {
|
||||
Op_ParanoidLoadMemTSO(IROp, ID);
|
||||
}
|
||||
else {
|
||||
Op_LoadMemTSO(IROp, ID);
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::IROps::OP_STOREMEMTSO:
|
||||
if (ParanoidTSO()) {
|
||||
Op_ParanoidStoreMemTSO(IROp, ID);
|
||||
}
|
||||
else {
|
||||
Op_StoreMemTSO(IROp, ID);
|
||||
}
|
||||
break;
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
@@ -1088,12 +1150,8 @@ void Arm64JITCore::ResetStack() {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::ContextImpl *CTX) {
|
||||
Arm64JITCore::InitializeSignalHandlers(CTX);
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return fextl::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
|
||||
+24
-29
@@ -6,7 +6,6 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
@@ -14,15 +13,18 @@ $end_info$
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -35,7 +37,7 @@ public:
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
@@ -48,8 +50,6 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
@@ -62,28 +62,7 @@ private:
|
||||
uint64_t Entry;
|
||||
CPUBackend::CompiledCode CodeData{};
|
||||
|
||||
std::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
constexpr static uint32_t NumGPRs = RA64.size();
|
||||
constexpr static uint32_t NumFPRs = RAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs = RA64Pair.size();
|
||||
constexpr static uint32_t NumCalleeGPRs = 10;
|
||||
constexpr static uint32_t NumCalleeGPRPairs = 5;
|
||||
constexpr static uint32_t RegisterCount = NumGPRs + NumFPRs + NumGPRPairs;
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
constexpr static uint64_t FPRBase = (1ULL << 32);
|
||||
constexpr static uint64_t GPRPairBase = (2ULL << 32);
|
||||
|
||||
/** @} */
|
||||
|
||||
constexpr static uint8_t RA_32 = 0;
|
||||
constexpr static uint8_t RA_64 = 1;
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
@@ -223,7 +202,7 @@ private:
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
@@ -231,6 +210,16 @@ private:
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
OpType RT_StoreRegister;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
@@ -318,6 +307,7 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -339,6 +329,8 @@ private:
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
@@ -349,7 +341,10 @@ private:
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(LoadMemTSO);
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(CacheLineClear);
|
||||
|
||||
+524
-2
@@ -128,6 +128,170 @@ DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ldrb(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldrh(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).W(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrb(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrh(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 8: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
strb(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
strh(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.W(), STATE, Op->Offset);
|
||||
break;
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs: {}", regOffs);
|
||||
strh(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
@@ -312,7 +476,7 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
DEF_OP(StoreRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -500,7 +664,6 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1182,6 +1345,106 @@ DEF_OP(LoadMemTSO) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorMasked) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Need SVE support in order to use VLoadVectorMasked");
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ld1h<ARMEmitter::SubRegSize::i16Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ld1w<ARMEmitter::SubRegSize::i32Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ld1d(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VLoadVectorMasked size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VStoreVectorMasked) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Need SVE support in order to use VStoreVectorMasked");
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VStoreVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto RegData = GetVReg(Op->Data.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemDst = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
st1h<ARMEmitter::SubRegSize::i16Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
st1w<ARMEmitter::SubRegSize::i32Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
st1d(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VStoreVectorMasked size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1503,6 +1766,265 @@ DEF_OP(MemSet) {
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
// TODO: A future looking task would be to support this with ARM's MOPS instructions.
|
||||
// The 8-bit non-atomic path directly matches ARM's CPYP/CPYM/CPYE instruction,
|
||||
//
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemRegDest = GetReg(Op->AddrDest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
|
||||
auto Dst = GetRegPair(Node);
|
||||
// If Direction == 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
// else:
|
||||
// MemRegDest is decremented (by size)
|
||||
// MemRegSrc is decremented (by size)
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::ForwardLabel BackwardImpl{};
|
||||
ARMEmitter::ForwardLabel Done{};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
if (Op->PrefixDest.IsInvalid()) {
|
||||
mov(TMP2, MemRegDest.X());
|
||||
}
|
||||
else {
|
||||
const auto Prefix = GetReg(Op->PrefixDest.ID());
|
||||
add(TMP2, Prefix.X(), MemRegDest.X());
|
||||
}
|
||||
|
||||
if (Op->PrefixSrc.IsInvalid()) {
|
||||
mov(TMP3, MemRegSrc.X());
|
||||
}
|
||||
else {
|
||||
const auto Prefix = GetReg(Op->PrefixSrc.ID());
|
||||
add(TMP3, Prefix.X(), MemRegSrc.X());
|
||||
}
|
||||
|
||||
// TMP1 = Length
|
||||
// TMP2 = Dest
|
||||
// TMP3 = Src
|
||||
// TMP4 = load+store temp value
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldrb<ARMEmitter::IndexType::POST>(TMP4.W(), TMP3, Size);
|
||||
strb<ARMEmitter::IndexType::POST>(TMP4.W(), TMP2, Size);
|
||||
break;
|
||||
case 2:
|
||||
ldrh<ARMEmitter::IndexType::POST>(TMP4.W(), TMP3, Size);
|
||||
strh<ARMEmitter::IndexType::POST>(TMP4.W(), TMP2, Size);
|
||||
break;
|
||||
case 4:
|
||||
ldr<ARMEmitter::IndexType::POST>(TMP4.W(), TMP3, Size);
|
||||
str<ARMEmitter::IndexType::POST>(TMP4.W(), TMP2, Size);
|
||||
break;
|
||||
case 8:
|
||||
ldr<ARMEmitter::IndexType::POST>(TMP4, TMP3, Size);
|
||||
str<ARMEmitter::IndexType::POST>(TMP4, TMP2, Size);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
auto MemCpyTSO = [this](uint32_t OpSize, int32_t Size) {
|
||||
if (CTX->HostFeatures.SupportsRCPC) {
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(TMP4.W(), TMP3);
|
||||
stlrb(TMP4.W(), TMP2);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(TMP4.W(), TMP3);
|
||||
break;
|
||||
case 4:
|
||||
ldapr(TMP4.W(), TMP3);
|
||||
break;
|
||||
case 8:
|
||||
ldapr(TMP4, TMP3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
stlrh(TMP4.W(), TMP2);
|
||||
break;
|
||||
case 4:
|
||||
stlr(TMP4.W(), TMP2);
|
||||
break;
|
||||
case 8:
|
||||
stlr(TMP4, TMP2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldarb(TMP4.W(), TMP3);
|
||||
stlrb(TMP4.W(), TMP2);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldarh(TMP4.W(), TMP3);
|
||||
break;
|
||||
case 4:
|
||||
ldar(TMP4.W(), TMP3);
|
||||
break;
|
||||
case 8:
|
||||
ldar(TMP4, TMP3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
stlrh(TMP4.W(), TMP2);
|
||||
break;
|
||||
case 4:
|
||||
stlr(TMP4.W(), TMP2);
|
||||
break;
|
||||
case 8:
|
||||
stlr(TMP4, TMP2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
if (Size >= 0) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, OpSize);
|
||||
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, OpSize);
|
||||
}
|
||||
else {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, OpSize);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, OpSize);
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
ARMEmitter::BackwardLabel AgainInternal{};
|
||||
ARMEmitter::ForwardLabel DoneInternal{};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
Bind(&AgainInternal);
|
||||
if (Op->IsAtomic) {
|
||||
MemCpyTSO(OpSize, SizeDirection);
|
||||
}
|
||||
else {
|
||||
MemCpy(OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
|
||||
// Needs to use temporaries just in case of overwrite
|
||||
mov(TMP1, MemRegDest.X());
|
||||
mov(TMP2, MemRegSrc.X());
|
||||
mov(TMP3, Length.X());
|
||||
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
add(Dst.first.X(), TMP1, TMP3);
|
||||
add(Dst.second.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
sub(Dst.first.X(), TMP1, TMP3);
|
||||
sub(Dst.second.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -4,11 +4,16 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <syscall.h>
|
||||
#endif
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
@@ -55,15 +60,15 @@ DEF_OP(Break) {
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
case Core::FAULT_SIGILL:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
case Core::FAULT_SIGTRAP:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
case Core::FAULT_SIGSEGV:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
|
||||
br(TMP1);
|
||||
break;
|
||||
@@ -139,7 +144,7 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
@@ -157,6 +162,7 @@ DEF_OP(Print) {
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(ProcessorID) {
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
@@ -164,7 +170,7 @@ DEF_OP(ProcessorID) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -207,6 +213,11 @@ DEF_OP(ProcessorID) {
|
||||
// Node is in w1
|
||||
orr(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ShiftType::LSL, 12);
|
||||
}
|
||||
#else
|
||||
DEF_OP(ProcessorID) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
+117
-54
@@ -361,7 +361,6 @@ DEF_OP(VAddP) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
|
||||
// SVE ADDP is a destructive operation, so we need a temporary
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
|
||||
// Unlike Adv. SIMD's version of ADDP, which acts like it concats the
|
||||
@@ -609,7 +608,6 @@ DEF_OP(VFAddP) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
|
||||
// SVE FADDP is a destructive operation, so we need a temporary
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
|
||||
// Unlike Adv. SIMD's version of FADDP, which acts like it concats the
|
||||
@@ -1533,17 +1531,13 @@ DEF_OP(VCMPEQ) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure no junk is in the temp (important for ensuring
|
||||
// non-equal entries remain as zero during the final bitwise OR).
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
|
||||
// General idea is to compare for equality, not the equal vals
|
||||
// from one of the registers, then or both together to make the
|
||||
// relevant equal entries all 1s.
|
||||
cmpeq(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmeq(SubRegSize.Scalar, Dst, Vector1, Vector2);
|
||||
@@ -1617,17 +1611,13 @@ DEF_OP(VCMPGT) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure no junk is in the temp (important for ensuring
|
||||
// non greater-than values remain as zero).
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
|
||||
// General idea is to compare for greater-than, bitwise NOT
|
||||
// the valid values, then ORR the NOTed values with the original
|
||||
// values to form entries that are all 1s.
|
||||
cmpgt(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmgt(SubRegSize.Scalar, Dst, Vector1, Vector2);
|
||||
@@ -1735,12 +1725,10 @@ DEF_OP(VFCMPEQ) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure we have no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
fcmeq(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -1784,12 +1772,10 @@ DEF_OP(VFCMPNEQ) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure we have no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
fcmne(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -1835,12 +1821,10 @@ DEF_OP(VFCMPLT) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure we have no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
fcmgt(SubRegSize.Vector, ComparePred, Mask, Vector2.Z(), Vector1.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector2.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector2.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector2.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -1884,12 +1868,10 @@ DEF_OP(VFCMPGT) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure there's no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
fcmgt(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -1933,12 +1915,10 @@ DEF_OP(VFCMPLE) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure there's no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
fcmge(SubRegSize.Vector, ComparePred, Mask, Vector2.Z(), Vector1.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector2.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector2.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector2.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -1983,17 +1963,14 @@ DEF_OP(VFCMPORD) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure there's no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
|
||||
// The idea is like comparing for unordered, but we just
|
||||
// invert the predicate from the comparison to instead
|
||||
// select all ordered elements in the vector.
|
||||
fcmuo(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(ComparePred, Mask, ComparePred);
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -2044,13 +2021,10 @@ DEF_OP(VFCMPUNO) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Ensure there's no junk in the temporary.
|
||||
eor(VTMP1.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
|
||||
fcmuo(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector1.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
switch (ElementSize) {
|
||||
@@ -2082,11 +2056,94 @@ DEF_OP(VFCMPUNO) {
|
||||
}
|
||||
|
||||
DEF_OP(VUShl) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
const auto Op = IROp->C<IR::IROp_VUShl>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto MaxShift = ElementSize * 8;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
dup_imm(SubRegSize, VTMP2.Z(), MaxShift);
|
||||
umin(SubRegSize, VTMP2.Z(), Mask, VTMP2.Z(), ShiftVector.Z());
|
||||
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP2.Z());
|
||||
} else {
|
||||
if (ElementSize < 8) {
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
|
||||
dup(SubRegSize, VTMP1.Q(), TMP1.R());
|
||||
|
||||
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
|
||||
cmhi(SubRegSize, VTMP2.Q(), ShiftVector.Q(), VTMP1.Q());
|
||||
bif(VTMP1.Q(), ShiftVector.Q(), VTMP2.Q());
|
||||
}
|
||||
|
||||
ushl(SubRegSize, Dst.Q(), Vector.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUShr) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
const auto Op = IROp->C<IR::IROp_VUShr>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto MaxShift = ElementSize * 8;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
dup_imm(SubRegSize, VTMP2.Z(), MaxShift);
|
||||
umin(SubRegSize, VTMP2.Z(), Mask, VTMP2.Z(), ShiftVector.Z());
|
||||
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP2.Z());
|
||||
} else {
|
||||
if (ElementSize < 8) {
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
|
||||
dup(SubRegSize, VTMP1.Q(), TMP1.R());
|
||||
|
||||
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
|
||||
cmhi(SubRegSize, VTMP2.Q(), ShiftVector.Q(), VTMP1.Q());
|
||||
bif(VTMP1.Q(), ShiftVector.Q(), VTMP2.Q());
|
||||
}
|
||||
|
||||
// Need to invert shift values to perform a right shift with USHL
|
||||
// (USHR only has an immediate variant).
|
||||
neg(SubRegSize, VTMP1.Q(), VTMP1.Q());
|
||||
ushl(SubRegSize, Dst.Q(), Vector.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSShr) {
|
||||
@@ -2111,17 +2168,23 @@ DEF_OP(VSShr) {
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
dup_imm(SubRegSize, VTMP2.Z(), MaxShift);
|
||||
umin(SubRegSize, VTMP2.Z(), Mask, VTMP2.Z(), ShiftVector.Z());
|
||||
dup_imm(SubRegSize, VTMP1.Z(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Z(), Mask, VTMP1.Z(), ShiftVector.Z());
|
||||
|
||||
movprfx(VTMP1.Z(), Vector.Z());
|
||||
asr(SubRegSize, VTMP1.Z(), Mask, VTMP1.Z(), VTMP2.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != 8, "Adv. SIMD UMIN doesn't handle 64-bit values");
|
||||
if (ElementSize < 8) {
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
|
||||
dup(SubRegSize, VTMP1.Q(), TMP1.R());
|
||||
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
|
||||
cmhi(SubRegSize, VTMP2.Q(), ShiftVector.Q(), VTMP1.Q());
|
||||
bif(VTMP1.Q(), ShiftVector.Q(), VTMP2.Q());
|
||||
}
|
||||
|
||||
// Need to invert shift values to perform a right shift with SSHL
|
||||
// (SSHR only has an immediate variant).
|
||||
|
||||
+4
-5
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
@@ -13,14 +14,12 @@ struct InternalThreadState;
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures();
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -5,6 +5,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -12,7 +13,6 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -12,7 +13,6 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
@@ -25,7 +26,6 @@ $end_info$
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
@@ -146,6 +146,7 @@ DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// XXX: This is very terrible, but I don't care for right now
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -186,7 +187,11 @@ DEF_OP(Syscall) {
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -302,6 +307,42 @@ DEF_OP(CPUID) {
|
||||
mov(Dst.second, rdx);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
|
||||
// CPUID ABI
|
||||
// this: rdi
|
||||
// Function: rsi
|
||||
//
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
mov(Dst.first.cvt32(), eax);
|
||||
mov(Dst.second, rax);
|
||||
shr(Dst.second, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -314,6 +355,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -5,13 +5,12 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
@@ -5,12 +5,11 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
@@ -5,12 +5,12 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
+70
-18
@@ -19,7 +19,6 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
@@ -28,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -35,12 +35,9 @@ $end_info$
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <signal.h>
|
||||
#include <sys/mman.h>
|
||||
#include <tuple>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
@@ -307,6 +304,66 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto SrcRAX = GetSrc<RA_64>(Op->RAX.ID());
|
||||
const auto SrcRDX = GetSrc<RA_64>(Op->RDX.ID());
|
||||
|
||||
// Encode the size check into the 8th bit to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(rdi, SrcRAX);
|
||||
mov(rsi, SrcRDX);
|
||||
|
||||
movq(rdx, LHS);
|
||||
pextrq(rcx, LHS, 1);
|
||||
|
||||
movq(r8, RHS);
|
||||
pextrq(r9, RHS, 1);
|
||||
|
||||
sub(rsp, 16);
|
||||
mov(dword [rsp], Control);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
add(rsp, 16);
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_32>(Node), rax);
|
||||
break;
|
||||
}
|
||||
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
movq(rdi, LHS);
|
||||
pextrq(rsi, LHS, 1);
|
||||
|
||||
movq(rdx, RHS);
|
||||
pextrq(rcx, RHS, 1);
|
||||
|
||||
mov(r8, Control);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_32>(Node), rax);
|
||||
break;
|
||||
}
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -349,7 +406,7 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses);
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, NumGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, NumXMMs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, NumGPRPairs);
|
||||
@@ -387,6 +444,11 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
@@ -399,12 +461,6 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return static_cast<Context::ContextImpl*>(Thread->CTX)->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
X86JITCore::~X86JITCore() {
|
||||
|
||||
}
|
||||
@@ -706,7 +762,7 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
if (IROp->Op != IR::OP_BEGINBLOCK &&
|
||||
IROp->Op != IR::OP_CONDJUMP &&
|
||||
IROp->Op != IR::OP_JUMP) {
|
||||
std::stringstream Inst;
|
||||
fextl::stringstream Inst;
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
|
||||
if (IROp->HasDest) {
|
||||
@@ -788,16 +844,12 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
return CodeData;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread);
|
||||
fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return fextl::make_unique<X86JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetX86JITBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::ContextImpl *CTX) {
|
||||
X86JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
}
|
||||
+13
-10
@@ -9,18 +9,20 @@ $end_info$
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
|
||||
using namespace Xbyak;
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <tuple>
|
||||
@@ -55,7 +57,7 @@ public:
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~X86JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
@@ -68,8 +70,6 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
@@ -123,7 +123,7 @@ private:
|
||||
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
@@ -140,7 +140,7 @@ private:
|
||||
uint64_t Entry;
|
||||
CPUBackend::CompiledCode CodeData{};
|
||||
|
||||
std::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
fextl::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
Xbyak::util::Cpu Features{};
|
||||
|
||||
bool MemoryDebug = false;
|
||||
@@ -151,7 +151,6 @@ private:
|
||||
constexpr static uint32_t NumGPRs = RA64.size(); // 4 is the minimum required for GPR ops
|
||||
constexpr static uint32_t NumXMMs = RAXMM.size();
|
||||
constexpr static uint32_t NumGPRPairs = RA64Pair.size();
|
||||
constexpr static uint32_t RegisterCount = NumGPRs + NumXMMs + NumGPRPairs;
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
@@ -314,6 +313,7 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -344,7 +344,10 @@ private:
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
+212
-2
@@ -6,7 +6,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -14,7 +14,6 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -766,6 +765,77 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Mask = GetSrc(Op->Mask.ID());
|
||||
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovps(ToYMM(Dst), ToYMM(Mask), yword [MemPtr]);
|
||||
} else {
|
||||
vmaskmovps(Dst, Mask, xword [MemPtr]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovpd(ToYMM(Dst), ToYMM(Mask), yword [MemPtr]);
|
||||
} else {
|
||||
vmaskmovpd(Dst, Mask, xword [MemPtr]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VLoadVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
DEF_OP(VStoreVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Data = GetDst(Op->Data.ID());
|
||||
const auto Mask = GetSrc(Op->Mask.ID());
|
||||
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovps(yword [MemPtr], ToYMM(Mask), ToYMM(Data));
|
||||
} else {
|
||||
vmaskmovps(xword [MemPtr], Mask, Data);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovpd(yword [MemPtr], ToYMM(Mask), ToYMM(Data));
|
||||
} else {
|
||||
vmaskmovpd(xword [MemPtr], Mask, Data);
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VStoreVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemSet) {
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
|
||||
@@ -791,6 +861,10 @@ DEF_OP(MemSet) {
|
||||
mov(rcx, Length);
|
||||
mov(rdi, MemReg);
|
||||
|
||||
if (!Op->Prefix.IsInvalid()) {
|
||||
add(rdi, GetSrc<RA_64>(Op->Prefix.ID()));
|
||||
}
|
||||
|
||||
{
|
||||
mov(TMP3, Length);
|
||||
auto CalculateDest = [&]() {
|
||||
@@ -854,6 +928,139 @@ DEF_OP(MemSet) {
|
||||
cld();
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemRegDest = GetSrc<RA_64>(Op->AddrDest.ID());
|
||||
const auto MemRegSrc = GetSrc<RA_64>(Op->AddrSrc.ID());
|
||||
|
||||
const auto Length = GetSrc<RA_64>(Op->Length.ID());
|
||||
const auto Direction = GetSrc<RA_64>(Op->Direction.ID());
|
||||
|
||||
// If Direction == 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
// else:
|
||||
// MemRegDest is decremented (by size)
|
||||
// MemRegSrc is decremented (by size)
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
// TMP1 = Length
|
||||
// TMP2 = Dest
|
||||
// TMP3 = Src
|
||||
// TMP4 = Temp value
|
||||
mov(TMP1, Length);
|
||||
mov(TMP2, MemRegDest);
|
||||
mov(TMP3, MemRegSrc);
|
||||
if (!Op->PrefixDest.IsInvalid()) {
|
||||
add(TMP2, GetSrc<RA_64>(Op->PrefixDest.ID()));
|
||||
}
|
||||
if (!Op->PrefixSrc.IsInvalid()) {
|
||||
add(TMP3, GetSrc<RA_64>(Op->PrefixSrc.ID()));
|
||||
}
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
Label Done;
|
||||
Label BackwardImpl;
|
||||
cmp(Direction, 0);
|
||||
jne(BackwardImpl);
|
||||
|
||||
// Emit forward direction memcpy then backward direction memcpy.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
Label DoneInternal;
|
||||
Label AgainInternal;
|
||||
|
||||
L(AgainInternal);
|
||||
cmp(TMP1, 0);
|
||||
je(DoneInternal);
|
||||
|
||||
{
|
||||
switch (Size) {
|
||||
case 1:
|
||||
movzx(TMP4, byte [TMP3]);
|
||||
mov(byte [TMP2], TMP4.cvt8());
|
||||
break;
|
||||
case 2:
|
||||
movzx(TMP4, word [TMP3]);
|
||||
mov(word [TMP2], TMP4.cvt16());
|
||||
break;
|
||||
case 4:
|
||||
mov(TMP4.cvt32(), dword [TMP3]);
|
||||
mov(dword [TMP2], TMP4.cvt32());
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP4, qword [TMP3]);
|
||||
mov(qword [TMP2], TMP4);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Direction == 1) {
|
||||
// Incrementing pointers
|
||||
add(TMP2, Size);
|
||||
add(TMP3, Size);
|
||||
}
|
||||
else {
|
||||
// Decrementing pointers
|
||||
sub(TMP2, Size);
|
||||
sub(TMP3, Size);
|
||||
}
|
||||
|
||||
// Decrement counter by one
|
||||
sub(TMP1, 1);
|
||||
|
||||
jmp(AgainInternal);
|
||||
L(DoneInternal);
|
||||
|
||||
// Pointer math using source pointers and length.
|
||||
mov(TMP3, Length);
|
||||
switch (Size) {
|
||||
case 1:
|
||||
break;
|
||||
case 2:
|
||||
shl(TMP3, 1);
|
||||
break;
|
||||
case 4:
|
||||
shl(TMP3, 2);
|
||||
break;
|
||||
case 8:
|
||||
shl(TMP3, 3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
|
||||
// Needs to use temporaries just in case of overwrite
|
||||
mov(TMP1, MemRegDest);
|
||||
mov(TMP2, MemRegSrc);
|
||||
|
||||
mov(Dst.first, TMP1);
|
||||
mov(Dst.second, TMP2);
|
||||
|
||||
if (Direction == 1) {
|
||||
// Incrementing pointers
|
||||
add(Dst.first, TMP3);
|
||||
add(Dst.second, TMP3);
|
||||
|
||||
jmp(Done);
|
||||
L(BackwardImpl);
|
||||
}
|
||||
else {
|
||||
// Decrementing pointers
|
||||
sub(Dst.first, TMP3);
|
||||
sub(Dst.second, TMP3);
|
||||
}
|
||||
}
|
||||
|
||||
L(Done);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
@@ -908,7 +1115,10 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
@@ -16,7 +17,6 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
@@ -43,6 +43,7 @@ DEF_OP(Fence) {
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
@@ -79,6 +80,11 @@ DEF_OP(Break) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
#else
|
||||
DEF_OP(Break) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
auto Dst = GetDst<RA_32>(Node);
|
||||
|
||||
@@ -5,7 +5,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
|
||||
@@ -5,14 +5,13 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -2726,11 +2725,71 @@ DEF_OP(VFCMPUNO) {
|
||||
}
|
||||
|
||||
DEF_OP(VUShl) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
const auto Op = IROp->C<IR::IROp_VUShl>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4 || ElementSize == 8,
|
||||
"VUShl only supports 32-bit and 64-bit elements");
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto ShiftVector = GetSrc(Op->ShiftVector.ID());
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
if (Is256Bit) {
|
||||
vpsllvd(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsllvd(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
case 8:
|
||||
if (Is256Bit) {
|
||||
vpsllvq(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsllvq(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUShr) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
const auto Op = IROp->C<IR::IROp_VUShr>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4 || ElementSize == 8,
|
||||
"VUShr only supports 32-bit and 64-bit elements");
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto ShiftVector = GetSrc(Op->ShiftVector.ID());
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
if (Is256Bit) {
|
||||
vpsrlvd(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsrlvd(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
case 8:
|
||||
if (Is256Bit) {
|
||||
vpsrlvq(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsrlvq(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSShr) {
|
||||
|
||||
+9
-11
@@ -11,15 +11,15 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore {
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
: ctx {CTX} {
|
||||
: BlockLinks_mbr { fextl::pmr::get_default_resource() }
|
||||
, ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma.new_object<BlockLinksMapType>();
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -33,7 +33,7 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
// Allocate a region of memory that we can use to back our block pointers
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, TotalCacheSize, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize));
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
// We need 32KB per guest page (One pointer per byte)
|
||||
@@ -52,7 +52,7 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
const size_t TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
|
||||
// No need to free BlockLinks map.
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
@@ -62,7 +62,7 @@ void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, MADV_DONTNEED);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
@@ -70,11 +70,9 @@ void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
madvise(reinterpret_cast<void*>(PagePointer), TotalCacheSize, MADV_DONTNEED);
|
||||
// Clear the BlockLinks allocator which frees the BlockLinks map implicitly.
|
||||
BlockLinks_mbr.release();
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma.new_object<BlockLinksMapType>();
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
BlockList.clear();
|
||||
}
|
||||
|
||||
+8
-10
@@ -1,22 +1,22 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <memory_resource>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
struct LookupCacheEntry {
|
||||
uintptr_t HostCode;
|
||||
uintptr_t GuestCode;
|
||||
@@ -67,7 +67,7 @@ public:
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
@@ -168,8 +168,6 @@ public:
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
@@ -245,10 +243,10 @@ private:
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
|
||||
std::pmr::polymorphic_allocator<std::byte> BlockLinks_pma {&BlockLinks_mbr};
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType *BlockLinks;
|
||||
|
||||
tsl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
|
||||
+21
-12
@@ -1,12 +1,16 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
struct
|
||||
FEX_PACKED
|
||||
CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie{};
|
||||
|
||||
// Instructions per block configuration
|
||||
@@ -16,41 +20,45 @@ namespace FEXCore::CodeSerialize {
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
bool MultiBlock : 1;
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
|
||||
// TSO enabled
|
||||
bool TSOEnabled : 1;
|
||||
unsigned TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
bool ABILocalFlags : 1;
|
||||
unsigned ABILocalFlags : 1;
|
||||
|
||||
// ABI no PF unsafe optimization
|
||||
bool ABINoPF : 1;
|
||||
unsigned ABINoPF : 1;
|
||||
|
||||
// Static register allocation enabled
|
||||
bool SRA : 1;
|
||||
unsigned SRA : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
bool ParanoidTSO : 1;
|
||||
unsigned ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
bool Is64BitMode : 1;
|
||||
unsigned Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
bool x87ReducedPrecision : 1;
|
||||
unsigned x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
unsigned _Pad : 17;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
@@ -67,6 +75,7 @@ namespace FEXCore::CodeSerialize {
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
@@ -79,6 +88,6 @@ namespace FEXCore::CodeSerialize {
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
|
||||
}
|
||||
@@ -2,25 +2,24 @@
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <sys/uio.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
|
||||
#ifndef _WIN32
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
auto BaseFilename = std::filesystem::path(filename).filename().string();
|
||||
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = std::make_unique<CodeRegionEntry>(
|
||||
auto Entry = fextl::make_unique<CodeRegionEntry>(
|
||||
Base,
|
||||
Size,
|
||||
Offset,
|
||||
@@ -77,12 +76,14 @@ namespace FEXCore::CodeSerialize {
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
#ifndef _WIN32
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
std::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
@@ -119,9 +120,10 @@ namespace FEXCore::CodeSerialize {
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data) {
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
}
|
||||
+6
-4
@@ -1,7 +1,9 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx) {
|
||||
@@ -23,14 +25,14 @@ namespace FEXCore::CodeSerialize {
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable) {
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
@@ -40,7 +42,7 @@ namespace FEXCore::CodeSerialize {
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <memory>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
@@ -38,13 +38,13 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, std::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
|
||||
if (Base == ~0ULL) {
|
||||
@@ -61,8 +61,7 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16] = "ObjectCodeSeri\0";
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
@@ -81,5 +80,5 @@ namespace FEXCore::CodeSerialize {
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,13 +6,14 @@
|
||||
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/queue.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <shared_mutex>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// XXX: Does this need to be signal safe?
|
||||
@@ -66,13 +67,13 @@ namespace FEXCore::CodeSerialize {
|
||||
uint64_t Offset{};
|
||||
|
||||
// Filename of the object
|
||||
std::string Filename{};
|
||||
fextl::string Filename{};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader{};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
std::string ObjectEntrySourceFilename{};
|
||||
fextl::string ObjectEntrySourceFilename{};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
@@ -105,12 +106,12 @@ namespace FEXCore::CodeSerialize {
|
||||
char *CodeData{};
|
||||
size_t FileSize{};
|
||||
|
||||
std::vector<CodeObjectFileSection> FileCodeSections;
|
||||
fextl::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
tsl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
|
||||
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
@@ -120,7 +121,7 @@ namespace FEXCore::CodeSerialize {
|
||||
CodeRegionEntry(uint64_t Base,
|
||||
uint64_t Size,
|
||||
uint64_t Offset,
|
||||
std::string const &Filename,
|
||||
fextl::string const &Filename,
|
||||
CodeObjectSerializationHeader const &DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
@@ -131,8 +132,8 @@ namespace FEXCore::CodeSerialize {
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = std::map<uint64_t, std::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = std::map<uint64_t, CodeRegionEntry*>;
|
||||
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
@@ -160,7 +161,7 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
@@ -186,9 +187,9 @@ namespace FEXCore::CodeSerialize {
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data);
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
@@ -219,22 +220,22 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const std::string &base, const std::string &filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
WorkItemAddNamedRegion(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry}
|
||||
{}
|
||||
const std::string BaseFilename;
|
||||
const std::string Filename;
|
||||
const fextl::string BaseFilename;
|
||||
const fextl::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, std::unique_ptr<CodeRegionEntry> entry)
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
@@ -242,7 +243,7 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
std::unique_ptr<CodeRegionEntry> Entry;
|
||||
fextl::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
@@ -281,9 +282,9 @@ namespace FEXCore::CodeSerialize {
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const std::string &base, const std::string &filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
void AsyncAddNamedRegionWorkItem(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(std::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
|
||||
base,
|
||||
filename,
|
||||
executable,
|
||||
@@ -292,9 +293,9 @@ namespace FEXCore::CodeSerialize {
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(std::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
|
||||
Base,
|
||||
Size,
|
||||
std::move(Entry)
|
||||
@@ -321,13 +322,13 @@ namespace FEXCore::CodeSerialize {
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
std::queue<std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
|
||||
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry);
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
|
||||
@@ -365,7 +366,7 @@ namespace FEXCore::CodeSerialize {
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
@@ -385,7 +386,7 @@ namespace FEXCore::CodeSerialize {
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(std::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
@@ -443,7 +444,7 @@ namespace FEXCore::CodeSerialize {
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
|
||||
Event WorkAvailable{};
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
+136
-91
@@ -7,12 +7,12 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
@@ -35,6 +35,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
constexpr size_t SyscallArgs = 7;
|
||||
using SyscallArray = std::array<uint64_t, SyscallArgs>;
|
||||
|
||||
size_t NumArguments{};
|
||||
const SyscallArray *GPRIndexes {};
|
||||
static constexpr SyscallArray GPRIndexes_64 = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
@@ -54,13 +55,26 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
FEXCore::X86State::REG_RDI,
|
||||
FEXCore::X86State::REG_RBP,
|
||||
};
|
||||
static_assert(GPRIndexes_64.size() == GPRIndexes_32.size());
|
||||
|
||||
static std::array<uint64_t, SyscallArgs> GPRIndexes_Hangover = {
|
||||
static constexpr SyscallArray GPRIndexes_Hangover = {
|
||||
FEXCore::X86State::REG_RCX,
|
||||
};
|
||||
|
||||
size_t NumArguments{};
|
||||
static constexpr SyscallArray GPRIndexes_Win64 = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_R10,
|
||||
FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_R8,
|
||||
FEXCore::X86State::REG_R9,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
};
|
||||
|
||||
static constexpr SyscallArray GPRIndexes_Win32 = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
};
|
||||
|
||||
SyscallFlags DefaultSyscallFlags = FEXCore::IR::SyscallFlags::DEFAULT;
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64) {
|
||||
@@ -68,9 +82,19 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
GPRIndexes = &GPRIndexes_64;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX32) {
|
||||
NumArguments = GPRIndexes_64.size();
|
||||
NumArguments = GPRIndexes_32.size();
|
||||
GPRIndexes = &GPRIndexes_32;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_WIN64) {
|
||||
NumArguments = 6;
|
||||
GPRIndexes = &GPRIndexes_Win64;
|
||||
DefaultSyscallFlags = FEXCore::IR::SyscallFlags::NORETURNEDRESULT;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_WIN32) {
|
||||
NumArguments = 2;
|
||||
GPRIndexes = &GPRIndexes_Win32;
|
||||
DefaultSyscallFlags = FEXCore::IR::SyscallFlags::NORETURNEDRESULT;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_HANGOVER) {
|
||||
NumArguments = 1;
|
||||
GPRIndexes = &GPRIndexes_Hangover;
|
||||
@@ -109,13 +133,20 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
Arguments[4],
|
||||
Arguments[5],
|
||||
Arguments[6],
|
||||
FEXCore::IR::SyscallFlags::DEFAULT);
|
||||
DefaultSyscallFlags);
|
||||
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_HANGOVER) {
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_HANGOVER &&
|
||||
(DefaultSyscallFlags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Hangover doesn't want us returning a result here
|
||||
// syscall is being abused as a thunk for now.
|
||||
StoreGPRRegister(X86State::REG_RAX, SyscallOp);
|
||||
}
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
_ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
@@ -433,7 +464,7 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs) {
|
||||
GenerateFlags_ADC(Op, Result, Before, Src, CF);
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
template<uint32_t SrcIndex, bool SetFlags>
|
||||
void OpDispatchBuilder::SBBOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
@@ -459,10 +490,12 @@ void OpDispatchBuilder::SBBOp(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
if (Size < 4) {
|
||||
Result = _Bfe(Size, Size * 8, 0, Result);
|
||||
if (SetFlags) {
|
||||
if (Size < 4) {
|
||||
Result = _Bfe(Size, Size * 8, 0, Result);
|
||||
}
|
||||
GenerateFlags_SBB(Op, Result, Before, Src, CF);
|
||||
}
|
||||
GenerateFlags_SBB(Op, Result, Before, Src, CF);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
|
||||
@@ -1511,7 +1544,7 @@ void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadGPRRegister(X86State::REG_RAX, 1, 8);
|
||||
|
||||
// Clear bits that aren't supposed to be set
|
||||
Src = _And(Src, _Constant(~0b101000));
|
||||
Src = _Andn(Src, _Constant(0b101000));
|
||||
|
||||
// Set the bit that is always set here
|
||||
Src = _Or(Src, _Constant(0b10));
|
||||
@@ -1746,6 +1779,18 @@ void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, _Bfe(32, 0, Result_Upper));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XGetBVOp(OpcodeArgs) {
|
||||
OrderedNode *Function = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
auto Res = _XGetBV(Function);
|
||||
|
||||
OrderedNode *Result_Lower = _ExtractElementPair(Res, 0);
|
||||
OrderedNode *Result_Upper = _ExtractElementPair(Res, 1);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, Result_Lower);
|
||||
StoreGPRRegister(X86State::REG_RDX, Result_Upper);
|
||||
}
|
||||
|
||||
template<bool SHL1Bit>
|
||||
void OpDispatchBuilder::SHLOp(OpcodeArgs) {
|
||||
OrderedNode *Src{};
|
||||
@@ -3805,7 +3850,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
OrderedNode *Counter = LoadGPRRegister(X86State::REG_RCX);
|
||||
auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC);
|
||||
|
||||
auto Result = _MemSet(CTX->IsTSOEnabled(), Size, Segment ?: InvalidNode, Dest, Src, Counter, DF);
|
||||
auto Result = _MemSet(CTX->IsAtomicTSOEnabled(), Size, Segment ?: InvalidNode, Dest, Src, Counter, DF);
|
||||
StoreGPRRegister(X86State::REG_RCX, _Constant(0));
|
||||
StoreGPRRegister(X86State::REG_RDI, Result);
|
||||
}
|
||||
@@ -3820,75 +3865,35 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
|
||||
// RA now can handle these to be here, to avoid DF accesses
|
||||
const auto Size = GetSrcSize(Op);
|
||||
auto SizeConst = _Constant(Size);
|
||||
auto NegSizeConst = _Constant(-Size);
|
||||
|
||||
// Calculate direction.
|
||||
auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC);
|
||||
auto PtrDir = _Select(FEXCore::IR::COND_EQ, DF, _Constant(0), SizeConst, NegSizeConst);
|
||||
|
||||
if (Op->Flags & (FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX | FEXCore::X86Tables::DecodeFlags::FLAG_REPNE_PREFIX)) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
auto SrcAddr = LoadGPRRegister(X86State::REG_RSI);
|
||||
auto DstAddr = LoadGPRRegister(X86State::REG_RDI);
|
||||
auto Counter = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
// Create all our blocks
|
||||
auto LoopHead = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto LoopTail = CreateNewCodeBlockAfter(LoopHead);
|
||||
auto LoopEnd = CreateNewCodeBlockAfter(LoopTail);
|
||||
auto DstSegment = GetSegment(0, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
auto SrcSegment = GetSegment(Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
auto Result = _MemCpy(CTX->IsAtomicTSOEnabled(), Size,
|
||||
DstSegment ?: InvalidNode,
|
||||
SrcSegment ?: InvalidNode,
|
||||
DstAddr, SrcAddr, Counter, DF);
|
||||
|
||||
// At the time this was written, our RA can't handle accessing nodes across blocks.
|
||||
// So we need to re-load and re-calculate essential values each iteration of the loop.
|
||||
OrderedNode *Result_Dst = _ExtractElementPair(Result, 0);
|
||||
OrderedNode *Result_Src = _ExtractElementPair(Result, 1);
|
||||
|
||||
// First thing we need to do is finish this block and jump to the start of the loop.
|
||||
|
||||
_Jump(LoopHead);
|
||||
|
||||
SetCurrentCodeBlock(LoopHead);
|
||||
{
|
||||
OrderedNode *Counter = LoadGPRRegister(X86State::REG_RCX);
|
||||
_CondJump(Counter, LoopEnd, LoopTail, {COND_EQ});
|
||||
}
|
||||
|
||||
SetCurrentCodeBlock(LoopTail);
|
||||
{
|
||||
OrderedNode *Src = LoadGPRRegister(X86State::REG_RSI);
|
||||
OrderedNode *Dest = LoadGPRRegister(X86State::REG_RDI);
|
||||
Dest = AppendSegmentOffset(Dest, 0, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
Src = AppendSegmentOffset(Src, Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
Src = _LoadMemAutoTSO(GPRClass, Size, Src, Size);
|
||||
|
||||
// Store to memory where RDI points
|
||||
_StoreMemAutoTSO(GPRClass, Size, Dest, Src, Size);
|
||||
|
||||
OrderedNode *TailCounter = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
// Decrement counter
|
||||
TailCounter = _Sub(TailCounter, _Constant(1));
|
||||
|
||||
// Store the counter so we don't have to deal with PHI here
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
|
||||
// Offset the pointer
|
||||
OrderedNode *TailSrc = LoadGPRRegister(X86State::REG_RSI);
|
||||
OrderedNode *TailDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
TailSrc = _Add(TailSrc, PtrDir);
|
||||
TailDest = _Add(TailDest, PtrDir);
|
||||
StoreGPRRegister(X86State::REG_RSI, TailSrc);
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest);
|
||||
|
||||
// Jump back to the start, we have more work to do
|
||||
_Jump(LoopHead);
|
||||
}
|
||||
|
||||
// Make sure to start a new block after ending this one
|
||||
|
||||
SetCurrentCodeBlock(LoopEnd);
|
||||
StoreGPRRegister(X86State::REG_RCX, _Constant(0));
|
||||
StoreGPRRegister(X86State::REG_RDI, Result_Dst);
|
||||
StoreGPRRegister(X86State::REG_RSI, Result_Src);
|
||||
}
|
||||
else {
|
||||
auto SizeConst = _Constant(Size);
|
||||
auto NegSizeConst = _Constant(-Size);
|
||||
auto PtrDir = _Select(FEXCore::IR::COND_EQ, DF, _Constant(0), SizeConst, NegSizeConst);
|
||||
|
||||
OrderedNode *RSI = LoadGPRRegister(X86State::REG_RSI);
|
||||
OrderedNode *RDI = LoadGPRRegister(X86State::REG_RDI);
|
||||
RDI= AppendSegmentOffset(RDI, 0, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
@@ -4684,7 +4689,7 @@ void OpDispatchBuilder::CMPXCHGPairOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CreateJumpBlocks(std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
void OpDispatchBuilder::CreateJumpBlocks(fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
OrderedNode *PrevCodeBlock{};
|
||||
for (auto &Target : *Blocks) {
|
||||
auto CodeNode = CreateCodeNode();
|
||||
@@ -4699,7 +4704,7 @@ void OpDispatchBuilder::CreateJumpBlocks(std::vector<FEXCore::Frontend::Decoder:
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
Entry = RIP;
|
||||
auto IRHeader = _IRHeader(InvalidNode, 0);
|
||||
Current_Header = IRHeader.first;
|
||||
@@ -5103,8 +5108,8 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
// TODO: Fix the instructions doing partial writes rather than dealing with it here.
|
||||
auto SrcVector = LoadXMMRegister(gprIndex);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Class != IR::GPRClass, "Partial writes from GPR not allowed. Instruction: {}",
|
||||
Op->TableInfo->Name);
|
||||
LOGMAN_THROW_A_FMT(Class != IR::GPRClass, "Partial writes from GPR not allowed. Instruction: {}",
|
||||
Op->TableInfo->Name);
|
||||
|
||||
// OpSize of 16 is special in that it is expected to zero the upper bits of the 256-bit operation.
|
||||
// TODO: Longer term we should enforce the difference between zero and insert.
|
||||
@@ -5355,14 +5360,19 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
case 0xCD: { // INT imm8
|
||||
uint8_t Literal = Op->Src[0].Data.Literal.Value;
|
||||
|
||||
if (Literal == 0x80) {
|
||||
#ifndef _WIN32
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x80;
|
||||
#else
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x2E;
|
||||
#endif
|
||||
if (Literal == SYSCALL_LITERAL) {
|
||||
// Syscall on linux
|
||||
SyscallOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Reason.ErrorRegister = Literal << 3 | (0b010);
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
// GP is raised when task-gate isn't setup to be valid
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
@@ -5370,33 +5380,33 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
}
|
||||
case 0xCE: // INTO
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_OF;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
case 0xF1: // INT1
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.Signal = Core::FAULT_SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_DB;
|
||||
Reason.si_code = 1;
|
||||
SetRIPToNext = true;
|
||||
break;
|
||||
case 0xF4: { // HLT
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
}
|
||||
case 0x0B: // UD2
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGILL;
|
||||
Reason.Signal = Core::FAULT_SIGILL;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_UD;
|
||||
Reason.si_code = 2;
|
||||
break;
|
||||
case 0xCC: // INT3
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.Signal = Core::FAULT_SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_BP;
|
||||
Reason.si_code = 0x80;
|
||||
SetRIPToNext = true;
|
||||
@@ -5709,6 +5719,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b00, 0x29), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPD_Op},
|
||||
{OPD(1, 0b01, 0x29), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPD_Op},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXCVTGPR_To_FPR<4>},
|
||||
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXCVTGPR_To_FPR<8>},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::VMOVVectorNTOp},
|
||||
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::VMOVVectorNTOp},
|
||||
|
||||
@@ -5759,6 +5772,11 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 4>},
|
||||
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5A), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4, true>},
|
||||
{OPD(1, 0b01, 0x5A), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8, true>},
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<8, 4>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<4, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<4, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<4, false, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<4, false, false>},
|
||||
@@ -5828,6 +5846,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4, true>},
|
||||
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::VPINSRWOp},
|
||||
{OPD(1, 0b01, 0xC5), 1, &OpDispatchBuilder::PExtrOp<2>},
|
||||
|
||||
{OPD(1, 0b00, 0xC6), 1, &OpDispatchBuilder::VSHUFOp<4>},
|
||||
@@ -5842,7 +5861,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b01, 0xD4), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 8>},
|
||||
{OPD(1, 0b01, 0xD5), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VSMUL, 2>},
|
||||
{OPD(1, 0b01, 0xD6), 1, &OpDispatchBuilder::MOVQOp},
|
||||
{OPD(1, 0b01, 0xD7), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(1, 0b01, 0xD7), 1, &OpDispatchBuilder::MOVMSKOpOne},
|
||||
|
||||
{OPD(1, 0b01, 0xD8), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VUQSUB, 1>},
|
||||
{OPD(1, 0b01, 0xD9), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VUQSUB, 2>},
|
||||
@@ -5881,6 +5900,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b01, 0xF3), 1, &OpDispatchBuilder::VPSLLOp<8>},
|
||||
{OPD(1, 0b01, 0xF4), 1, &OpDispatchBuilder::VPMULLOp<4, false>},
|
||||
{OPD(1, 0b01, 0xF5), 1, &OpDispatchBuilder::VPMADDWDOp},
|
||||
{OPD(1, 0b01, 0xF6), 1, &OpDispatchBuilder::VPSADBWOp},
|
||||
{OPD(1, 0b01, 0xF7), 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
|
||||
{OPD(1, 0b01, 0xF8), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VSUB, 1>},
|
||||
@@ -5895,6 +5915,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x01), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, 2>},
|
||||
{OPD(2, 0b01, 0x02), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, 4>},
|
||||
{OPD(2, 0b01, 0x03), 1, &OpDispatchBuilder::VPHADDSWOp},
|
||||
{OPD(2, 0b01, 0x04), 1, &OpDispatchBuilder::VPMADDUBSWOp},
|
||||
|
||||
{OPD(2, 0b01, 0x05), 1, &OpDispatchBuilder::VPHSUBOp<2>},
|
||||
{OPD(2, 0b01, 0x06), 1, &OpDispatchBuilder::VPHSUBOp<4>},
|
||||
@@ -5906,6 +5927,8 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x0B), 1, &OpDispatchBuilder::VPMULHRSWOp},
|
||||
{OPD(2, 0b01, 0x0C), 1, &OpDispatchBuilder::VPERMILRegOp<4>},
|
||||
{OPD(2, 0b01, 0x0D), 1, &OpDispatchBuilder::VPERMILRegOp<8>},
|
||||
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::VTESTPOp<4>},
|
||||
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::VTESTPOp<8>},
|
||||
|
||||
{OPD(2, 0b01, 0x16), 1, &OpDispatchBuilder::VPERMDOp},
|
||||
{OPD(2, 0b01, 0x17), 1, &OpDispatchBuilder::PTestOp},
|
||||
@@ -5927,6 +5950,10 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x29), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VCMPEQ, 8>},
|
||||
{OPD(2, 0b01, 0x2A), 1, &OpDispatchBuilder::VMOVVectorNTOp},
|
||||
{OPD(2, 0b01, 0x2B), 1, &OpDispatchBuilder::VPACKUSOp<4>},
|
||||
{OPD(2, 0b01, 0x2C), 1, &OpDispatchBuilder::VMASKMOVOp<4, false>},
|
||||
{OPD(2, 0b01, 0x2D), 1, &OpDispatchBuilder::VMASKMOVOp<8, false>},
|
||||
{OPD(2, 0b01, 0x2E), 1, &OpDispatchBuilder::VMASKMOVOp<4, true>},
|
||||
{OPD(2, 0b01, 0x2F), 1, &OpDispatchBuilder::VMASKMOVOp<8, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x30), 1, &OpDispatchBuilder::AVXExtendVectorElements<1, 2, false>},
|
||||
{OPD(2, 0b01, 0x31), 1, &OpDispatchBuilder::AVXExtendVectorElements<1, 4, false>},
|
||||
@@ -5948,7 +5975,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
|
||||
{OPD(2, 0b01, 0x40), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VSMUL, 4>},
|
||||
{OPD(2, 0b01, 0x41), 1, &OpDispatchBuilder::VPHMINPOSUWOp},
|
||||
{OPD(2, 0b01, 0x45), 1, &OpDispatchBuilder::VPSRLVOp},
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::VBROADCASTOp<4>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::VBROADCASTOp<8>},
|
||||
@@ -5957,6 +5986,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x78), 1, &OpDispatchBuilder::VBROADCASTOp<1>},
|
||||
{OPD(2, 0b01, 0x79), 1, &OpDispatchBuilder::VBROADCASTOp<2>},
|
||||
|
||||
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::VPMASKMOVOp<false>},
|
||||
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::VPMASKMOVOp<true>},
|
||||
|
||||
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::VAESIMCOp},
|
||||
{OPD(2, 0b01, 0xDC), 1, &OpDispatchBuilder::VAESEncOp},
|
||||
{OPD(2, 0b01, 0xDD), 1, &OpDispatchBuilder::VAESEncLastOp},
|
||||
@@ -5985,14 +6017,16 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
|
||||
{OPD(3, 0b01, 0x18), 1, &OpDispatchBuilder::VINSERTOp},
|
||||
{OPD(3, 0b01, 0x19), 1, &OpDispatchBuilder::VEXTRACT128Op},
|
||||
|
||||
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::VPINSRBOp},
|
||||
{OPD(3, 0b01, 0x21), 1, &OpDispatchBuilder::VINSERTPSOp},
|
||||
{OPD(3, 0b01, 0x22), 1, &OpDispatchBuilder::VPINSRDQOp},
|
||||
|
||||
{OPD(3, 0b01, 0x38), 1, &OpDispatchBuilder::VINSERTOp},
|
||||
{OPD(3, 0b01, 0x39), 1, &OpDispatchBuilder::VEXTRACT128Op},
|
||||
|
||||
{OPD(3, 0b01, 0x40), 1, &OpDispatchBuilder::VDPPOp<4>},
|
||||
{OPD(3, 0b01, 0x41), 1, &OpDispatchBuilder::VDPPOp<8>},
|
||||
{OPD(3, 0b01, 0x42), 1, &OpDispatchBuilder::VMPSADBWOp},
|
||||
|
||||
{OPD(3, 0b01, 0x46), 1, &OpDispatchBuilder::VPERM2Op},
|
||||
|
||||
@@ -6000,6 +6034,11 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(3, 0b01, 0x4B), 1, &OpDispatchBuilder::AVXVectorVariableBlend<8>},
|
||||
{OPD(3, 0b01, 0x4C), 1, &OpDispatchBuilder::AVXVectorVariableBlend<1>},
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::VAESKeyGenAssistOp},
|
||||
};
|
||||
#undef OPD
|
||||
@@ -6075,7 +6114,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
{0x10, 6, &OpDispatchBuilder::ADCOp<0>},
|
||||
|
||||
{0x18, 6, &OpDispatchBuilder::SBBOp<0>},
|
||||
{0x18, 6, &OpDispatchBuilder::SBBOp<0, true>},
|
||||
|
||||
{0x20, 6, &OpDispatchBuilder::ALUOp<FEXCore::IR::IROps::OP_AND, FEXCore::IR::IROps::OP_ATOMICFETCHAND, false>},
|
||||
|
||||
@@ -6157,6 +6196,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xCE, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xD4, 1, &OpDispatchBuilder::AAMOp},
|
||||
{0xD5, 1, &OpDispatchBuilder::AADOp},
|
||||
{0xD6, 1, &OpDispatchBuilder::SBBOp<0, false>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> BaseOpTable_64[] = {
|
||||
@@ -6227,7 +6267,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x57, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VXOR, 16>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFADD, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFMUL, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFSUB, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFMIN, 4>},
|
||||
@@ -6318,7 +6358,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::ADCOp<1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::SBBOp<1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::SBBOp<1, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
@@ -6327,7 +6367,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::ADCOp<1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::SBBOp<1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::SBBOp<1, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
@@ -6336,7 +6376,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::ADCOp<1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::SBBOp<1>}, // Unit tests find this setting flags incorrectly
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::SBBOp<1, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
@@ -6516,7 +6556,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x57, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VXOR, 16>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFADD, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFMUL, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFSUB, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFMIN, 8>},
|
||||
@@ -6708,7 +6748,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryModRMExtensionOpTable[] = {
|
||||
// REG /2
|
||||
{((1 << 3) | 0), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{((1 << 3) | 0), 1, &OpDispatchBuilder::XGetBVOp},
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
@@ -7287,6 +7327,11 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
#undef PF_3A_NONE
|
||||
|
||||
+77
-10
@@ -1,24 +1,24 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <fmt/format.h>
|
||||
#include <map>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
@@ -85,7 +85,7 @@ public:
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
std::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
|
||||
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
@@ -157,7 +157,7 @@ public:
|
||||
bool HadDecodeFailure() const { return DecodeFailure; }
|
||||
bool NeedsBlockEnder() const { return NeedsBlockEnd; }
|
||||
|
||||
void BeginFunction(uint64_t RIP, std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void Finalize();
|
||||
|
||||
// Dispatch builder functions
|
||||
@@ -181,7 +181,7 @@ public:
|
||||
void SecondaryALUOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void ADCOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
template<uint32_t SrcIndex, bool SetFlags>
|
||||
void SBBOp(OpcodeArgs);
|
||||
void PUSHOp(OpcodeArgs);
|
||||
void PUSHREGOp(OpcodeArgs);
|
||||
@@ -219,6 +219,7 @@ public:
|
||||
void MOVOffsetOp(OpcodeArgs);
|
||||
void CMOVOp(OpcodeArgs);
|
||||
void CPUIDOp(OpcodeArgs);
|
||||
void XGetBVOp(OpcodeArgs);
|
||||
template<bool SHL1Bit>
|
||||
void SHLOp(OpcodeArgs);
|
||||
void SHLImmediateOp(OpcodeArgs);
|
||||
@@ -355,7 +356,7 @@ public:
|
||||
void Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void Scalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
template<size_t DstElementSize, size_t SrcElementSize, bool IsAVX>
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
@@ -415,12 +416,18 @@ public:
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
void AVXVectorRound(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template <size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void AVXVector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template <size_t SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize>
|
||||
void AVXCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
void AVXVFCMPOp(OpcodeArgs);
|
||||
|
||||
@@ -456,6 +463,9 @@ public:
|
||||
void VINSERTOp(OpcodeArgs);
|
||||
void VINSERTPSOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool IsStore>
|
||||
void VMASKMOVOp(OpcodeArgs);
|
||||
|
||||
void VMOVAPS_VMOVAPD_Op(OpcodeArgs);
|
||||
void VMOVUPS_VMOVUPD_Op(OpcodeArgs);
|
||||
|
||||
@@ -471,6 +481,8 @@ public:
|
||||
|
||||
void VMOVVectorNTOp(OpcodeArgs);
|
||||
|
||||
void VMPSADBWOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPACKSSOp(OpcodeArgs);
|
||||
|
||||
@@ -479,6 +491,11 @@ public:
|
||||
|
||||
void VPALIGNROp(OpcodeArgs);
|
||||
|
||||
void VPCMPESTRIOp(OpcodeArgs);
|
||||
void VPCMPESTRMOp(OpcodeArgs);
|
||||
void VPCMPISTRIOp(OpcodeArgs);
|
||||
void VPCMPISTRMOp(OpcodeArgs);
|
||||
|
||||
void VPERM2Op(OpcodeArgs);
|
||||
void VPERMDOp(OpcodeArgs);
|
||||
void VPERMQOp(OpcodeArgs);
|
||||
@@ -496,8 +513,16 @@ public:
|
||||
void VPHSUBOp(OpcodeArgs);
|
||||
void VPHSUBSWOp(OpcodeArgs);
|
||||
|
||||
void VPINSRBOp(OpcodeArgs);
|
||||
void VPINSRDQOp(OpcodeArgs);
|
||||
void VPINSRWOp(OpcodeArgs);
|
||||
|
||||
void VPMADDUBSWOp(OpcodeArgs);
|
||||
void VPMADDWDOp(OpcodeArgs);
|
||||
|
||||
template <bool IsStore>
|
||||
void VPMASKMOVOp(OpcodeArgs);
|
||||
|
||||
void VPMULHRSWOp(OpcodeArgs);
|
||||
|
||||
template <bool Signed>
|
||||
@@ -506,6 +531,8 @@ public:
|
||||
template <size_t ElementSize, bool Signed>
|
||||
void VPMULLOp(OpcodeArgs);
|
||||
|
||||
void VPSADBWOp(OpcodeArgs);
|
||||
|
||||
void VPSHUFBOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Low>
|
||||
@@ -516,6 +543,7 @@ public:
|
||||
void VPSLLDQOp(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void VPSLLIOp(OpcodeArgs);
|
||||
void VPSLLVOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSRAOp(OpcodeArgs);
|
||||
@@ -524,6 +552,7 @@ public:
|
||||
void VPSRAIOp(OpcodeArgs);
|
||||
|
||||
void VPSRAVDOp(OpcodeArgs);
|
||||
void VPSRLVOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VPSRLDOp(OpcodeArgs);
|
||||
@@ -541,6 +570,9 @@ public:
|
||||
template <size_t ElementSize>
|
||||
void VSHUFOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void VTESTPOp(OpcodeArgs);
|
||||
|
||||
void VZEROOp(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
@@ -795,9 +827,15 @@ private:
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorVariableBlend(OpcodeArgs);
|
||||
|
||||
void AVXVariableShiftImpl(OpcodeArgs, IROps IROp);
|
||||
|
||||
OrderedNode* AESKeyGenAssistImpl(OpcodeArgs);
|
||||
OrderedNode* AESIMCImpl(OpcodeArgs);
|
||||
|
||||
OrderedNode* CVTGPR_To_FPRImpl(OpcodeArgs, size_t DstElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
@@ -813,6 +851,10 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* MPSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& ImmOp);
|
||||
|
||||
OrderedNode* PACKSSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
@@ -823,6 +865,8 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
OrderedNode* PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2);
|
||||
|
||||
@@ -834,9 +878,17 @@ private:
|
||||
OrderedNode* PHSUBSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* PINSROpImpl(OpcodeArgs, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* PMADDWDOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2);
|
||||
|
||||
OrderedNode* PMADDUBSWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* PMULHRSWOpImpl(OpcodeArgs, OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PMULHWOpImpl(OpcodeArgs, bool Signed,
|
||||
@@ -845,6 +897,9 @@ private:
|
||||
OrderedNode* PMULLOpImpl(OpcodeArgs, size_t ElementSize, bool Signed,
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
OrderedNode* PSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* PSHUFBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2);
|
||||
|
||||
@@ -868,11 +923,17 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
void VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore,
|
||||
const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp);
|
||||
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
|
||||
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
@@ -882,6 +943,12 @@ private:
|
||||
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Mode);
|
||||
|
||||
OrderedNode* Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, bool IsAVX);
|
||||
|
||||
OrderedNode* Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
@@ -1484,21 +1551,21 @@ private:
|
||||
return !Op->Dest.IsGPR();
|
||||
}
|
||||
|
||||
void CreateJumpBlocks(std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void CreateJumpBlocks(fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
bool BlockSetRIP {false};
|
||||
|
||||
bool Multiblock{};
|
||||
uint64_t Entry;
|
||||
|
||||
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
if (CTX->IsTSOEnabled())
|
||||
if (CTX->IsAtomicTSOEnabled())
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
if (CTX->IsTSOEnabled())
|
||||
if (CTX->IsAtomicTSOEnabled())
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
@@ -5,7 +5,8 @@ desc: Handles x86/64 Crypto instructions to IR
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
@@ -77,7 +78,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
using FnType = OrderedNode* (*)(OpDispatchBuilder&, OrderedNode*, OrderedNode*, OrderedNode*);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(Self._And(B, C), Self._And(Self._Not(B), D));
|
||||
return Self._Xor(Self._And(B, C), Self._Andn(D, B));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(Self._Xor(B, C), D);
|
||||
@@ -204,7 +205,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
|
||||
return _Xor(_And(E, F), _And(_Not(E), G));
|
||||
return _Xor(_And(E, F), _Andn(G, E));
|
||||
};
|
||||
const auto Major = [this](OrderedNode *A, OrderedNode *B, OrderedNode *C) -> OrderedNode* {
|
||||
return _Xor(_Xor(_And(A, B), _And(A, C)), _And(B, C));
|
||||
|
||||
@@ -7,10 +7,10 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
@@ -52,10 +52,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
auto Tmp = _And(_Lshr(Src, _Constant(FlagOffset)), OneConst);
|
||||
auto Tmp = _Bfe(4, 1, FlagOffset, Src);
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
}
|
||||
}
|
||||
@@ -304,28 +303,11 @@ void OpDispatchBuilder::CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, O
|
||||
// OF
|
||||
// Signed
|
||||
{
|
||||
auto NegOne = _Constant(~0ULL);
|
||||
auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne);
|
||||
auto XorOp1 = _Not(_Xor(Src1, Src2));
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (Size) {
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
case 16:
|
||||
AndOp1 = _Bfe(1, 15, AndOp1);
|
||||
break;
|
||||
case 32:
|
||||
AndOp1 = _Bfe(1, 31, AndOp1);
|
||||
break;
|
||||
case 64:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", Size);
|
||||
break;
|
||||
}
|
||||
AndOp1 = _Bfe(1, Size - 1, AndOp1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
@@ -376,24 +358,7 @@ void OpDispatchBuilder::CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, O
|
||||
auto XorOp1 = _Xor(Src1, Src2);
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (SrcSize) {
|
||||
case 1:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
case 2:
|
||||
AndOp1 = _Bfe(1, 15, AndOp1);
|
||||
break;
|
||||
case 4:
|
||||
AndOp1 = _Bfe(1, 31, AndOp1);
|
||||
break;
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", SrcSize);
|
||||
break;
|
||||
}
|
||||
AndOp1 = _Bfe(1, SrcSize * 8 - 1, AndOp1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
@@ -492,29 +457,12 @@ void OpDispatchBuilder::CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, O
|
||||
|
||||
// OF
|
||||
{
|
||||
auto NegOne = _Constant(~0ULL);
|
||||
auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne);
|
||||
auto XorOp1 = _Not(_Xor(Src1, Src2));
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (SrcSize) {
|
||||
case 1:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
case 2:
|
||||
AndOp1 = _Bfe(1, 15, AndOp1);
|
||||
break;
|
||||
case 4:
|
||||
AndOp1 = _Bfe(1, 31, AndOp1);
|
||||
break;
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", SrcSize);
|
||||
break;
|
||||
}
|
||||
AndOp1 = _Bfe(1, SrcSize * 8 - 1, AndOp1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
|
||||
+564
-208
@@ -7,11 +7,11 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
@@ -847,7 +847,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) {
|
||||
|
||||
for (unsigned i = 0; i < NumElements; ++i) {
|
||||
// Extract the top bit of the element
|
||||
OrderedNode *Tmp = _VExtractToGPR(16, ElementSize, Src, i);
|
||||
OrderedNode *Tmp = _VExtractToGPR(Size, ElementSize, Src, i);
|
||||
Tmp = _Bfe(1, ElementSize * 8 - 1, Tmp);
|
||||
|
||||
// Shift it to the correct location
|
||||
@@ -865,17 +865,28 @@ template
|
||||
void OpDispatchBuilder::MOVMSKOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ExtractSize = Is256Bit ? 4 : 2;
|
||||
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *VMask = _VDupFromGPR(16, 8, _Constant(0x80'40'20'10'08'04'02'01ULL));
|
||||
OrderedNode *VMask = _VDupFromGPR(SrcSize, 8, _Constant(0x80'40'20'10'08'04'02'01ULL));
|
||||
|
||||
auto VCMP = _VCMPLTZ(16, 1, Src);
|
||||
auto VAnd = _VAnd(16, 1, VCMP, VMask);
|
||||
auto VCMP = _VCMPLTZ(SrcSize, 1, Src);
|
||||
auto VAnd = _VAnd(SrcSize, 1, VCMP, VMask);
|
||||
|
||||
auto VAdd1 = _VAddP(16, 1, VAnd, VAnd);
|
||||
auto VAdd2 = _VAddP(8, 1, VAdd1, VAdd1);
|
||||
// Since we also handle the MM MOVMSKB here too,
|
||||
// we need to clamp the lower bound.
|
||||
const auto VAdd1Size = std::max(SrcSize, uint8_t{16});
|
||||
const auto VAdd2Size = std::max(SrcSize / 2, 8);
|
||||
|
||||
auto VAdd1 = _VAddP(VAdd1Size, 1, VAnd, VAnd);
|
||||
auto VAdd2 = _VAddP(VAdd2Size, 1, VAdd1, VAdd1);
|
||||
auto VAdd3 = _VAddP(8, 1, VAdd2, VAdd2);
|
||||
|
||||
StoreResult(GPRClass, Op, _VExtractToGPR(16, 2, VAdd3, 0), -1);
|
||||
auto Result = _VExtractToGPR(SrcSize, ExtractSize, VAdd3, 0);
|
||||
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
@@ -1276,28 +1287,32 @@ void OpDispatchBuilder::VBROADCASTOp<8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VBROADCASTOp<16>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PINSROpImpl(OpcodeArgs, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
const auto NumElements = Size / ElementSize;
|
||||
|
||||
OrderedNode *Src2{};
|
||||
if (Src2Op.IsGPR()) {
|
||||
Src2 = LoadSource(GPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
} else {
|
||||
// If loading from memory then we only load the element size
|
||||
Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, ElementSize, Op->Flags, -1);
|
||||
}
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, Size, Op->Flags, -1);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be literal here");
|
||||
const uint64_t Index = Imm.Data.Literal.Value & (NumElements - 1);
|
||||
|
||||
return _VInsGPR(Size, ElementSize, Index, Src1, Src2);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PINSROp(OpcodeArgs) {
|
||||
auto Size = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src{};
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
}
|
||||
else {
|
||||
// If loading from memory then we only load the element size
|
||||
Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ElementSize, Op->Flags, -1);
|
||||
}
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
uint64_t Index = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
uint8_t NumElements = Size / ElementSize;
|
||||
Index &= NumElements - 1;
|
||||
|
||||
// This maps 1:1 to an AArch64 NEON Op
|
||||
auto ALUOp = _VInsGPR(Size, ElementSize, Index, Dest, Src);
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
OrderedNode *Result = PINSROpImpl(Op, ElementSize, Op->Dest, Op->Src[0], Op->Src[1]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
@@ -1309,6 +1324,28 @@ void OpDispatchBuilder::PINSROp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::PINSROp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VPINSRBOp(OpcodeArgs) {
|
||||
OrderedNode *Result = PINSROpImpl(Op, 1, Op->Src[0], Op->Src[1], Op->Src[2]);
|
||||
OrderedNode *Final = _VMov(16, Result);
|
||||
|
||||
StoreResult(FPRClass, Op, Final, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPINSRDQOp(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
OrderedNode *Result = PINSROpImpl(Op, SrcSize, Op->Src[0], Op->Src[1], Op->Src[2]);
|
||||
OrderedNode *Final = _VMov(16, Result);
|
||||
|
||||
StoreResult(FPRClass, Op, Final, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPINSRWOp(OpcodeArgs) {
|
||||
OrderedNode *Result = PINSROpImpl(Op, 2, Op->Src[0], Op->Src[1], Op->Src[2]);
|
||||
OrderedNode *Final = _VMov(16, Result);
|
||||
|
||||
StoreResult(FPRClass, Op, Final, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
@@ -1854,13 +1891,19 @@ void OpDispatchBuilder::VPSRAIOp<2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPSRAIOp<4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VPSRAVDOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVXVariableShiftImpl(OpcodeArgs, IROps IROp) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Vector = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *ShiftVector = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Result = _VSShr(SrcSize, 4, Vector, ShiftVector);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Vector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
|
||||
OrderedNode *ShiftVector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], DstSize, Op->Flags, -1);
|
||||
|
||||
auto Shift = _VUShr(DstSize, SrcSize, Vector, ShiftVector);
|
||||
Shift.first->Header.Op = IROp;
|
||||
|
||||
OrderedNode *Result = Shift;
|
||||
if (Is128Bit) {
|
||||
Result = _VMov(16, Result);
|
||||
}
|
||||
@@ -1868,6 +1911,18 @@ void OpDispatchBuilder::VPSRAVDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPSLLVOp(OpcodeArgs) {
|
||||
AVXVariableShiftImpl(Op, IROps::OP_VUSHL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPSRAVDOp(OpcodeArgs) {
|
||||
AVXVariableShiftImpl(Op, IROps::OP_VSSHR);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPSRLVOp(OpcodeArgs) {
|
||||
AVXVariableShiftImpl(Op, IROps::OP_VUSHR);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Res = _VDupElement(16, GetSrcSize(Op), Src, 0);
|
||||
@@ -1895,19 +1950,22 @@ void OpDispatchBuilder::VMOVDDUPOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::CVTGPR_To_FPRImpl(OpcodeArgs, size_t DstElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto GPRSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, 16, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(GPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
OrderedNode *Converted = _Float_FromGPR_S(DstElementSize, GPRSize, Src2);
|
||||
|
||||
return _VInsElement(16, DstElementSize, 0, 0, Src1, Converted);
|
||||
}
|
||||
|
||||
template<size_t DstElementSize>
|
||||
void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
size_t GPRSize = GetSrcSize(Op);
|
||||
|
||||
Src = _Float_FromGPR_S(DstElementSize, GPRSize, Src);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
OrderedNode *Result = CVTGPR_To_FPRImpl(Op, DstElementSize, Op->Dest, Op->Src[0]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
@@ -1915,6 +1973,16 @@ void OpDispatchBuilder::CVTGPR_To_FPR<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::CVTGPR_To_FPR<8>(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize>
|
||||
void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs) {
|
||||
OrderedNode *Result = CVTGPR_To_FPRImpl(Op, DstElementSize, Op->Src[0], Op->Src[1]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template
|
||||
void OpDispatchBuilder::AVXCVTGPR_To_FPR<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXCVTGPR_To_FPR<8>(OpcodeArgs);
|
||||
|
||||
template<size_t SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
@@ -2058,17 +2126,20 @@ void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<8, true, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<8, true, true>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, 16, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcElementSize, Op->Flags, -1);
|
||||
|
||||
OrderedNode *Converted = _Float_FToF(DstElementSize, SrcElementSize, Src2);
|
||||
|
||||
return _VInsElement(16, DstElementSize, 0, 0, Src1, Converted);
|
||||
}
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
Src = _Float_FToF(DstElementSize, SrcElementSize, Src);
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
auto Result = _VInsElement(DstSize, DstElementSize, 0, 0, Dest, Src);
|
||||
OrderedNode *Result = Scalar_CVT_Float_To_FloatImpl(Op, DstElementSize, SrcElementSize, Op->Dest, Op->Src[0]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -2077,26 +2148,55 @@ void OpDispatchBuilder::Scalar_CVT_Float_To_Float<4, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::Scalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
if constexpr (DstElementSize > SrcElementSize) {
|
||||
Src = _Vector_FToF(Size, SrcElementSize << 1, Src, SrcElementSize);
|
||||
}
|
||||
else {
|
||||
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
template <size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::AVXScalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
OrderedNode *Result = Scalar_CVT_Float_To_FloatImpl(Op, DstElementSize, SrcElementSize, Op->Src[0], Op->Src[1]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<4, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, bool IsAVX) {
|
||||
const auto IsFloatSrc = SrcElementSize == 4;
|
||||
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto StoreSize = IsFloatSrc ? SrcSize
|
||||
: 16;
|
||||
const auto LoadSize = IsFloatSrc ? SrcSize / 2
|
||||
: SrcSize;
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags, -1);
|
||||
|
||||
OrderedNode *Result{};
|
||||
if (DstElementSize > SrcElementSize) {
|
||||
Result = _Vector_FToF(SrcSize, SrcElementSize << 1, Src, SrcElementSize);
|
||||
} else {
|
||||
Result = _Vector_FToF(SrcSize, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
}
|
||||
|
||||
if (IsAVX && GetDstSize(Op) == 16) {
|
||||
Result = _VMov(16, Result);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, StoreSize, -1);
|
||||
}
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize, bool IsAVX>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
Vector_CVT_Float_To_FloatImpl(Op, DstElementSize, SrcElementSize, IsAVX);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4, true>(OpcodeArgs);
|
||||
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
@@ -2193,6 +2293,50 @@ void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore,
|
||||
const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp) {
|
||||
|
||||
const auto MakeAddress = [this, Op](const X86Tables::DecodedOperand& Data) {
|
||||
OrderedNode *BaseAddr = LoadSource_WithOpSize(GPRClass, Op, Data, CTX->GetGPRSize(), Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(BaseAddr, Op->Flags);
|
||||
};
|
||||
|
||||
OrderedNode *Mask = LoadSource_WithOpSize(FPRClass, Op, MaskOp, DataSize, Op->Flags, -1);
|
||||
|
||||
if (IsStore) {
|
||||
OrderedNode *Data = LoadSource_WithOpSize(FPRClass, Op, DataOp, DataSize, Op->Flags, -1);
|
||||
OrderedNode *Address = MakeAddress(Op->Dest);
|
||||
_VStoreVectorMasked(DataSize, ElementSize, Mask, Data, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
OrderedNode *Address = MakeAddress(DataOp);
|
||||
OrderedNode *Result = _VLoadVectorMasked(DataSize, ElementSize, Mask, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
}
|
||||
|
||||
template <size_t ElementSize, bool IsStore>
|
||||
void OpDispatchBuilder::VMASKMOVOp(OpcodeArgs) {
|
||||
VMASKMOVOpImpl(Op, ElementSize, GetDstSize(Op), IsStore, Op->Src[0], Op->Src[1]);
|
||||
}
|
||||
template
|
||||
void OpDispatchBuilder::VMASKMOVOp<4, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VMASKMOVOp<4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VMASKMOVOp<8, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VMASKMOVOp<8, true>(OpcodeArgs);
|
||||
|
||||
template <bool IsStore>
|
||||
void OpDispatchBuilder::VPMASKMOVOp(OpcodeArgs) {
|
||||
VMASKMOVOpImpl(Op, GetSrcSize(Op), GetDstSize(Op), IsStore, Op->Src[0], Op->Src[1]);
|
||||
}
|
||||
template
|
||||
void OpDispatchBuilder::VPMASKMOVOp<false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPMASKMOVOp<true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR() &&
|
||||
Op->Dest.Data.GPR.GPR >= FEXCore::X86State::REG_XMM_0) {
|
||||
@@ -2959,20 +3103,12 @@ void OpDispatchBuilder::VPMADDWDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) {
|
||||
// This is a pretty curious operation
|
||||
// Does four MADD operations across 8 8bit signed and unsigned integers and accumulates to 16bit integers in the destination WITH saturation
|
||||
//
|
||||
// x86 PMADDUBSW: mm1, mm2
|
||||
// mm1[15:0] = SaturateSigned16(((s8)mm2[15:8] * (u8)mm1[15:8]) + ((s8)mm2[7:0] * (u8)mm1[7:0]))
|
||||
// mm1[31:16] = SaturateSigned16(((s8)mm2[31:24] * (u8)mm1[31:24]) + ((s8)mm2[23:16] * (u8)mm1[23:16]))
|
||||
// mm1[47:32] = SaturateSigned16(((s8)mm2[47:40] * (u8)mm1[47:40]) + ((s8)mm2[39:32] * (u8)mm1[39:32]))
|
||||
// mm1[63:48] = SaturateSigned16(((s8)mm2[63:56] * (u8)mm1[63:56]) + ((s8)mm2[55:48] * (u8)mm1[55:48]))
|
||||
// Extends to larger registers
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode* OpDispatchBuilder::PMADDUBSWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
// 64bit is more efficient
|
||||
@@ -2990,34 +3126,40 @@ void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) {
|
||||
auto ResAdd = _VAddP(Size * 2, 4, ResMul_L, ResMul_H);
|
||||
|
||||
// Add saturate back down to 16bit
|
||||
OrderedNode *Res = _VSQXTN(Size * 2, 4, ResAdd);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
return _VSQXTN(Size * 2, 4, ResAdd);
|
||||
}
|
||||
else {
|
||||
// Src1 is unsigned
|
||||
auto Src1_16b_L = _VUXTL(Size, 1, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
auto Src1_16b_H = _VUXTL2(Size, 1, Src1); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
|
||||
// Src2 is signed
|
||||
auto Src2_16b_L = _VSXTL(Size, 1, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
auto Src2_16b_H = _VSXTL2(Size, 1, Src2); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
// Src1 is unsigned
|
||||
auto Src1_16b_L = _VUXTL(Size, 1, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
auto Src1_16b_H = _VUXTL2(Size, 1, Src1); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
|
||||
auto ResMul_L = _VSMull(Size, 2, Src1_16b_L, Src2_16b_L);
|
||||
auto ResMul_L_H = _VSMull2(Size, 2, Src1_16b_L, Src2_16b_L);
|
||||
// Src2 is signed
|
||||
auto Src2_16b_L = _VSXTL(Size, 1, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
auto Src2_16b_H = _VSXTL2(Size, 1, Src2); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56]
|
||||
|
||||
auto ResMul_H = _VSMull(Size, 2, Src1_16b_H, Src2_16b_H);
|
||||
auto ResMul_H_H = _VSMull2(Size, 2, Src1_16b_H, Src2_16b_H);
|
||||
auto ResMul_L = _VSMull(Size, 2, Src1_16b_L, Src2_16b_L);
|
||||
auto ResMul_L_H = _VSMull2(Size, 2, Src1_16b_L, Src2_16b_L);
|
||||
|
||||
// Now add pairwise across the vector
|
||||
auto ResAdd_L = _VAddP(Size, 4, ResMul_L, ResMul_L_H);
|
||||
auto ResAdd_H = _VAddP(Size, 4, ResMul_H, ResMul_H_H);
|
||||
auto ResMul_H = _VSMull(Size, 2, Src1_16b_H, Src2_16b_H);
|
||||
auto ResMul_H_H = _VSMull2(Size, 2, Src1_16b_H, Src2_16b_H);
|
||||
|
||||
// Add saturate back down to 16bit
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, ResAdd_L);
|
||||
Res = _VSQXTN2(Size, 4, Res, ResAdd_H);
|
||||
// Now add pairwise across the vector
|
||||
auto ResAdd_L = _VAddP(Size, 4, ResMul_L, ResMul_L_H);
|
||||
auto ResAdd_H = _VAddP(Size, 4, ResMul_H, ResMul_H_H);
|
||||
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
// Add saturate back down to 16bit
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, ResAdd_L);
|
||||
return _VSQXTN2(Size, 4, Res, ResAdd_H);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) {
|
||||
OrderedNode * Result = PMADDUBSWOpImpl(Op, Op->Dest, Op->Src[0]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) {
|
||||
OrderedNode * Result = PMADDUBSWOpImpl(Op, Op->Src[0], Op->Src[1]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed,
|
||||
@@ -3397,49 +3539,72 @@ void OpDispatchBuilder::VPHSUBSWOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PSADBW(OpcodeArgs) {
|
||||
OrderedNode* OpDispatchBuilder::PSADBWOpImpl(OpcodeArgs,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
// The documentation is actually incorrect in how this instruction operates
|
||||
// It strongly implies that the `abs(dest[i] - src[i])` operates in 8bit space
|
||||
// but it actually operates in more than 8bit space
|
||||
// This can be seen with `abs(0 - 0xFF)` returning a different result depending
|
||||
// on bit length
|
||||
auto Size = GetSrcSize(Op);
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
OrderedNode *Result{};
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
Dest = _VUXTL(Size*2, 1, Dest);
|
||||
Src = _VUXTL(Size*2, 1, Src);
|
||||
OrderedNode *Src1_Low = _VUXTL(Size*2, 1, Src1);
|
||||
OrderedNode *Src2_Low = _VUXTL(Size*2, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult = _VSub(Size*2, 2, Dest, Src);
|
||||
OrderedNode *SubResult = _VSub(Size*2, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *AbsResult = _VAbs(Size*2, 2, SubResult);
|
||||
|
||||
// Now vector-wide add the results for each
|
||||
Result = _VAddV(Size * 2, 2, AbsResult);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Dest_Low = _VUXTL(Size, 1, Dest);
|
||||
OrderedNode *Dest_High = _VUXTL2(Size, 1, Dest);
|
||||
|
||||
OrderedNode *Src_Low = _VUXTL(Size, 1, Src);
|
||||
OrderedNode *Src_High = _VUXTL2(Size, 1, Src);
|
||||
|
||||
OrderedNode *SubResult_Low = _VSub(Size, 2, Dest_Low, Src_Low);
|
||||
OrderedNode *SubResult_High = _VSub(Size, 2, Dest_High, Src_High);
|
||||
|
||||
OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low);
|
||||
OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High);
|
||||
|
||||
// Now vector pairwise add all four of these
|
||||
OrderedNode * Result_Low = _VAddV(Size, 2, AbsResult_Low);
|
||||
OrderedNode * Result_High = _VAddV(Size, 2, AbsResult_High);
|
||||
|
||||
Result = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High);
|
||||
return _VAddV(Size * 2, 2, AbsResult);
|
||||
}
|
||||
|
||||
|
||||
OrderedNode *Src1_Low = _VUXTL(Size, 1, Src1);
|
||||
OrderedNode *Src1_High = _VUXTL2(Size, 1, Src1);
|
||||
|
||||
OrderedNode *Src2_Low = _VUXTL(Size, 1, Src2);
|
||||
OrderedNode *Src2_High = _VUXTL2(Size, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult_Low = _VSub(Size, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *SubResult_High = _VSub(Size, 2, Src1_High, Src2_High);
|
||||
|
||||
OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low);
|
||||
OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High);
|
||||
|
||||
OrderedNode *Result_Low = _VAddV(16, 2, AbsResult_Low);
|
||||
OrderedNode *Result_High = _VAddV(16, 2, AbsResult_High);
|
||||
|
||||
OrderedNode *Low = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High);
|
||||
if (Is128Bit) {
|
||||
return Low;
|
||||
}
|
||||
|
||||
OrderedNode *HighSrc1 = _VDupElement(Size, 16, AbsResult_Low, 1);
|
||||
OrderedNode *HighSrc2 = _VDupElement(Size, 16, AbsResult_High, 1);
|
||||
|
||||
OrderedNode *HighResult_Low = _VAddV(16, 2, HighSrc1);
|
||||
OrderedNode *HighResult_High = _VAddV(16, 2, HighSrc2);
|
||||
|
||||
OrderedNode *High = _VInsElement(Size, 8, 1, 0, HighResult_Low, HighResult_High);
|
||||
OrderedNode *Full = _VInsElement(Size, 16, 1, 0, Low, High);
|
||||
|
||||
OrderedNode *Tmp = _VInsElement(Size, 8, 2, 1, Full, Full);
|
||||
return _VInsElement(Size, 8, 1, 2, Tmp, Full);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PSADBW(OpcodeArgs) {
|
||||
OrderedNode *Result = PSADBWOpImpl(Op, Op->Dest, Op->Src[0]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPSADBWOp(OpcodeArgs) {
|
||||
OrderedNode *Result = PSADBWOpImpl(Op, Op->Src[0], Op->Src[1]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -3764,6 +3929,59 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(ZeroConst);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VTESTOpImpl(OpcodeArgs, size_t ElementSize) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto ElementSizeInBits = ElementSize * 8;
|
||||
const auto MaskConstant = uint64_t{1} << (ElementSizeInBits - 1);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
OrderedNode *Mask = _VDupFromGPR(SrcSize, ElementSize, _Constant(MaskConstant));
|
||||
|
||||
OrderedNode *AndTest = _VAnd(SrcSize, 1, Src2, Src1);
|
||||
OrderedNode *AndNotTest = _VBic(SrcSize, 1, Src2, Src1);
|
||||
|
||||
OrderedNode *MaskedAnd = _VAnd(SrcSize, 1, AndTest, Mask);
|
||||
OrderedNode *MaskedAndNot = _VAnd(SrcSize, 1, AndNotTest, Mask);
|
||||
|
||||
OrderedNode *AndPopCount = _VPopcount(SrcSize, 1, MaskedAnd);
|
||||
OrderedNode *AndNotPopCount = _VPopcount(SrcSize, 1, MaskedAndNot);
|
||||
|
||||
OrderedNode *SummedAnd = _VAddV(SrcSize, 2, AndPopCount);
|
||||
OrderedNode *SummedAndNot = _VAddV(SrcSize, 2, AndNotPopCount);
|
||||
|
||||
OrderedNode *AndGPR = _VExtractToGPR(SrcSize, 2, SummedAnd, 0);
|
||||
OrderedNode *AndNotGPR = _VExtractToGPR(SrcSize, 2, SummedAndNot, 0);
|
||||
|
||||
OrderedNode *ZeroConst = _Constant(0);
|
||||
OrderedNode *OneConst = _Constant(1);
|
||||
|
||||
OrderedNode *ZFResult = _Select(IR::COND_EQ, AndGPR, ZeroConst,
|
||||
OneConst, ZeroConst);
|
||||
OrderedNode *CFResult = _Select(IR::COND_EQ, AndNotGPR, ZeroConst,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFResult);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(ZeroConst);
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(ZeroConst);
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(ZeroConst);
|
||||
}
|
||||
|
||||
template <size_t ElementSize>
|
||||
void OpDispatchBuilder::VTESTPOp(OpcodeArgs) {
|
||||
VTESTOpImpl(Op, ElementSize);
|
||||
}
|
||||
template
|
||||
void OpDispatchBuilder::VTESTPOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VTESTPOp<8>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PHMINPOSUWOpImpl(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
@@ -3887,90 +4105,118 @@ void OpDispatchBuilder::VDPPOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VDPPOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
uint8_t Select = Op->Src[1].Data.Literal.Value;
|
||||
OrderedNode* OpDispatchBuilder::MPSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& ImmOp) {
|
||||
const auto LaneHelper = [&, this](uint32_t Selector_Src1, uint32_t Selector_Src2,
|
||||
OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Src2 will grab a 32bit element and duplicate it across the 128bits
|
||||
OrderedNode *DupSrc = _VDupElement(16, 4, Src2, Selector_Src2);
|
||||
|
||||
// Src1/Dest needs a bunch of magic
|
||||
|
||||
// Shift right by selected bytes
|
||||
// This will give us Dest[15:0], and Dest[79:64]
|
||||
OrderedNode *Dest1 = _VExtr(16, 1, Src1, Src1, Selector_Src1 + 0);
|
||||
// This will give us Dest[31:16], and Dest[95:80]
|
||||
OrderedNode *Dest2 = _VExtr(16, 1, Src1, Src1, Selector_Src1 + 1);
|
||||
// This will give us Dest[47:32], and Dest[111:96]
|
||||
OrderedNode *Dest3 = _VExtr(16, 1, Src1, Src1, Selector_Src1 + 2);
|
||||
// This will give us Dest[63:48], and Dest[127:112]
|
||||
OrderedNode *Dest4 = _VExtr(16, 1, Src1, Src1, Selector_Src1 + 3);
|
||||
|
||||
// For each shifted section, we now have two 32-bit values per vector that can be used
|
||||
// Dest1.S[0] and Dest1.S[1] = Bytes - 0,1,2,3:4,5,6,7
|
||||
// Dest2.S[0] and Dest2.S[1] = Bytes - 1,2,3,4:5,6,7,8
|
||||
// Dest3.S[0] and Dest3.S[1] = Bytes - 2,3,4,5:6,7,8,9
|
||||
// Dest4.S[0] and Dest4.S[1] = Bytes - 3,4,5,6:7,8,9,10
|
||||
Dest1 = _VUABDL(16, 1, Dest1, DupSrc);
|
||||
Dest2 = _VUABDL(16, 1, Dest2, DupSrc);
|
||||
Dest3 = _VUABDL(16, 1, Dest3, DupSrc);
|
||||
Dest4 = _VUABDL(16, 1, Dest4, DupSrc);
|
||||
|
||||
// Dest[1,2,3,4] Now contains the data prior to combining
|
||||
// Temp[0,1,2,3] for each step
|
||||
|
||||
// Each destination now has 16bit x 8 elements in it that were the absolute difference for each byte
|
||||
// Needs each to be 16bit to store the next step
|
||||
// Next stage is to sum pairwise
|
||||
// Dest1:
|
||||
// ADDP Dest2, Dest1: TmpCombine1
|
||||
// ADDP Dest4, Dest3: TmpCombine2
|
||||
// TmpCombine1.8H[0] = Dest1.8H[0] + Dest1.8H[1];
|
||||
// TmpCombine1.8H[1] = Dest1.8H[2] + Dest1.8H[3];
|
||||
// TmpCombine1.8H[2] = Dest1.8H[4] + Dest1.8H[5];
|
||||
// TmpCombine1.8H[3] = Dest1.8H[6] + Dest1.8H[7];
|
||||
// TmpCombine1.8H[4] = Dest2.8H[0] + Dest2.8H[1];
|
||||
// TmpCombine1.8H[5] = Dest2.8H[2] + Dest2.8H[3];
|
||||
// TmpCombine1.8H[6] = Dest2.8H[4] + Dest2.8H[5];
|
||||
// TmpCombine1.8H[7] = Dest2.8H[6] + Dest2.8H[7];
|
||||
// <Repeat for Dest4 and Dest3>
|
||||
// ADDP TmpCombine2, TmpCombine1: FinalCombine
|
||||
// FinalCombine.8H[0] = TmpCombine1.8H[0] + TmpCombine1.8H[1]
|
||||
// FinalCombine.8H[1] = TmpCombine1.8H[2] + TmpCombine1.8H[3]
|
||||
// FinalCombine.8H[2] = TmpCombine1.8H[4] + TmpCombine1.8H[5]
|
||||
// FinalCombine.8H[3] = TmpCombine1.8H[6] + TmpCombine1.8H[7]
|
||||
// FinalCombine.8H[4] = TmpCombine2.8H[0] + TmpCombine2.8H[1]
|
||||
// FinalCombine.8H[5] = TmpCombine2.8H[2] + TmpCombine2.8H[3]
|
||||
// FinalCombine.8H[6] = TmpCombine2.8H[4] + TmpCombine2.8H[5]
|
||||
// FinalCombine.8H[7] = TmpCombine2.8H[6] + TmpCombine2.8H[7]
|
||||
|
||||
auto TmpCombine1 = _VAddP(16, 2, Dest1, Dest2);
|
||||
auto TmpCombine2 = _VAddP(16, 2, Dest3, Dest4);
|
||||
|
||||
auto FinalCombine = _VAddP(16, 2, TmpCombine1, TmpCombine2);
|
||||
|
||||
// This now contains our results but they are in the wrong order.
|
||||
// We need to swizzle the results in to the correct ordering
|
||||
// Result.8H[0] = FinalCombine.8H[0]
|
||||
// Result.8H[1] = FinalCombine.8H[2]
|
||||
// Result.8H[2] = FinalCombine.8H[4]
|
||||
// Result.8H[3] = FinalCombine.8H[6]
|
||||
// Result.8H[4] = FinalCombine.8H[1]
|
||||
// Result.8H[5] = FinalCombine.8H[3]
|
||||
// Result.8H[6] = FinalCombine.8H[5]
|
||||
// Result.8H[7] = FinalCombine.8H[7]
|
||||
|
||||
auto Even = _VUnZip(16, 2, FinalCombine, FinalCombine);
|
||||
auto Odd = _VUnZip2(16, 2, FinalCombine, FinalCombine);
|
||||
return _VInsElement(16, 8, 1, 0, Even, Odd);
|
||||
};
|
||||
|
||||
LOGMAN_THROW_A_FMT(ImmOp.IsLiteral(), "ImmOp needs to be literal here");
|
||||
const uint8_t Select = ImmOp.Data.Literal.Value;
|
||||
const uint8_t SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
// Src1 needs to be in byte offset
|
||||
uint8_t Select_Dest = ((Select & 0b100) >> 2) * 32 / 8;
|
||||
uint8_t Select_Src2 = Select & 0b11;
|
||||
const uint8_t Select_Src1_Low = ((Select & 0b100) >> 2) * 32 / 8;
|
||||
const uint8_t Select_Src2_Low = Select & 0b11;
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
// Src2 will grab a 32bit element and duplicate it across the 128bits
|
||||
OrderedNode *DupSrc = _VDupElement(16, 4, Src, Select_Src2);
|
||||
OrderedNode *Lower = LaneHelper(Select_Src1_Low, Select_Src2_Low, Src1, Src2);
|
||||
if (Is128Bit) {
|
||||
return Lower;
|
||||
}
|
||||
|
||||
// Src1/Dest needs a bunch of magic
|
||||
const uint8_t Select_Src1_High = ((Select & 0b100000) >> 5) * 32 / 8;
|
||||
const uint8_t Select_Src2_High = (Select & 0b11000) >> 3;
|
||||
|
||||
// Shift right by selected bytes
|
||||
// This will give us Dest[15:0], and Dest[79:64]
|
||||
OrderedNode *Dest1 = _VExtr(16, 1, Dest, Dest, Select_Dest + 0);
|
||||
// This will give us Dest[31:16], and Dest[95:80]
|
||||
OrderedNode *Dest2 = _VExtr(16, 1, Dest, Dest, Select_Dest + 1);
|
||||
// This will give us Dest[47:32], and Dest[111:96]
|
||||
OrderedNode *Dest3 = _VExtr(16, 1, Dest, Dest, Select_Dest + 2);
|
||||
// This will give us Dest[63:48], and Dest[127:112]
|
||||
OrderedNode *Dest4 = _VExtr(16, 1, Dest, Dest, Select_Dest + 3);
|
||||
OrderedNode *UpperSrc1 = _VDupElement(32, 16, Src1, 1);
|
||||
OrderedNode *UpperSrc2 = _VDupElement(32, 16, Src2, 1);
|
||||
OrderedNode *Upper = LaneHelper(Select_Src1_High, Select_Src2_High, UpperSrc1, UpperSrc2);
|
||||
return _VInsElement(32, 16, 1, 0, Lower, Upper);
|
||||
}
|
||||
|
||||
// For each shifted section, we now have two 32-bit values per vector that can be used
|
||||
// Dest1.S[0] and Dest1.S[1] = Bytes - 0,1,2,3:4,5,6,7
|
||||
// Dest2.S[0] and Dest2.S[1] = Bytes - 1,2,3,4:5,6,7,8
|
||||
// Dest3.S[0] and Dest3.S[1] = Bytes - 2,3,4,5:6,7,8,9
|
||||
// Dest4.S[0] and Dest4.S[1] = Bytes - 3,4,5,6:7,8,9,10
|
||||
Dest1 = _VUABDL(16, 1, Dest1, DupSrc);
|
||||
Dest2 = _VUABDL(16, 1, Dest2, DupSrc);
|
||||
Dest3 = _VUABDL(16, 1, Dest3, DupSrc);
|
||||
Dest4 = _VUABDL(16, 1, Dest4, DupSrc);
|
||||
|
||||
// Dest[1,2,3,4] Now contains the data prior to combining
|
||||
// Temp[0,1,2,3] for each step
|
||||
|
||||
// Each destination now has 16bit x 8 elements in it that were the absolute difference for each byte
|
||||
// Needs each to be 16bit to store the next step
|
||||
// Next stage is to sum pairwise
|
||||
// Dest1:
|
||||
// ADDP Dest2, Dest1: TmpCombine1
|
||||
// ADDP Dest4, Dest3: TmpCombine2
|
||||
// TmpCombine1.8H[0] = Dest1.8H[0] + Dest1.8H[1];
|
||||
// TmpCombine1.8H[1] = Dest1.8H[2] + Dest1.8H[3];
|
||||
// TmpCombine1.8H[2] = Dest1.8H[4] + Dest1.8H[5];
|
||||
// TmpCombine1.8H[3] = Dest1.8H[6] + Dest1.8H[7];
|
||||
// TmpCombine1.8H[4] = Dest2.8H[0] + Dest2.8H[1];
|
||||
// TmpCombine1.8H[5] = Dest2.8H[2] + Dest2.8H[3];
|
||||
// TmpCombine1.8H[6] = Dest2.8H[4] + Dest2.8H[5];
|
||||
// TmpCombine1.8H[7] = Dest2.8H[6] + Dest2.8H[7];
|
||||
// <Repeat for Dest4 and Dest3>
|
||||
// ADDP TmpCombine2, TmpCombine1: FinalCombine
|
||||
// FinalCombine.8H[0] = TmpCombine1.8H[0] + TmpCombine1.8H[1]
|
||||
// FinalCombine.8H[1] = TmpCombine1.8H[2] + TmpCombine1.8H[3]
|
||||
// FinalCombine.8H[2] = TmpCombine1.8H[4] + TmpCombine1.8H[5]
|
||||
// FinalCombine.8H[3] = TmpCombine1.8H[6] + TmpCombine1.8H[7]
|
||||
// FinalCombine.8H[4] = TmpCombine2.8H[0] + TmpCombine2.8H[1]
|
||||
// FinalCombine.8H[5] = TmpCombine2.8H[2] + TmpCombine2.8H[3]
|
||||
// FinalCombine.8H[6] = TmpCombine2.8H[4] + TmpCombine2.8H[5]
|
||||
// FinalCombine.8H[7] = TmpCombine2.8H[6] + TmpCombine2.8H[7]
|
||||
|
||||
auto TmpCombine1 = _VAddP(16, 2, Dest1, Dest2);
|
||||
auto TmpCombine2 = _VAddP(16, 2, Dest3, Dest4);
|
||||
|
||||
auto FinalCombine = _VAddP(16, 2, TmpCombine1, TmpCombine2);
|
||||
|
||||
// This now contains our results but they are in the wrong order.
|
||||
// We need to swizzle the results in to the correct ordering
|
||||
// Result.8H[0] = FinalCombine.8H[0]
|
||||
// Result.8H[1] = FinalCombine.8H[2]
|
||||
// Result.8H[2] = FinalCombine.8H[4]
|
||||
// Result.8H[3] = FinalCombine.8H[6]
|
||||
// Result.8H[4] = FinalCombine.8H[1]
|
||||
// Result.8H[5] = FinalCombine.8H[3]
|
||||
// Result.8H[6] = FinalCombine.8H[5]
|
||||
// Result.8H[7] = FinalCombine.8H[7]
|
||||
|
||||
auto Even = _VUnZip(16, 2, FinalCombine, FinalCombine);
|
||||
auto Odd = _VUnZip2(16, 2, FinalCombine, FinalCombine);
|
||||
auto Result = _VInsElement(16, 8, 1, 0, Even, Odd);
|
||||
void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) {
|
||||
OrderedNode *Result = MPSADBWOpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMPSADBWOp(OpcodeArgs) {
|
||||
OrderedNode *Result = MPSADBWOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -4345,4 +4591,114 @@ void OpDispatchBuilder::VPERMILRegOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPERMILRegOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[1] needs to be a literal");
|
||||
const auto Control = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
// SSE4.2 string instructions modify flags, so invalidate
|
||||
// any previously deferred flags.
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
// NOTE: Unlike most other SSE/AVX instructions, the SSE4.2 string and text
|
||||
// instructions do *not* require memory operands to be aligned on a 16 byte
|
||||
// boundary (see "Other Exceptions" descriptions for the relevant
|
||||
// instructions in the Intel Software Development Manual).
|
||||
//
|
||||
// So, we specify Src2 as having an alignment of 1 to indicate this.
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, 1);
|
||||
|
||||
OrderedNode *IntermediateResult{};
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(SrcSize, Src1, Src2, SrcRAX, SrcRDX, Control);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
|
||||
OrderedNode *ZeroConst = _Constant(0);
|
||||
|
||||
if (IsMask) {
|
||||
// For the masked variant of the instructions, if control[6] is set, then we
|
||||
// need to expand the intermediate result into a byte or word mask (depending
|
||||
// on data size specified in control[1]) along the entire length of XMM0,
|
||||
// where set bits in the intermediate result set the corresponding entry
|
||||
// in XMM0 to all 1s and unset bits set the corresponding entry to all 0s.
|
||||
//
|
||||
// If control[6] is not set, then we just store the intermediate result as-is
|
||||
// into the least significant bits of XMM0 and zero extend it.
|
||||
const auto IsExpandedMask = (Control & 0b0100'0000) != 0;
|
||||
|
||||
if (IsExpandedMask) {
|
||||
// We need to iterate over the intermediate result and
|
||||
// expand the mask into XMM0 elements.
|
||||
const auto ElementSize = 1U << (Control & 1);
|
||||
const auto NumElements = 16U >> (Control & 1);
|
||||
|
||||
OrderedNode *Result = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
OrderedNode *SignBit = _Sbfe(1, i, IntermediateResult);
|
||||
Result = _VInsGPR(Core::CPUState::XMM_SSE_REG_SIZE, ElementSize, i, Result, SignBit);
|
||||
}
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
// We insert the intermediate result as-is.
|
||||
StoreXMMRegister(0, _VCastFromGPR(16, 2, IntermediateResult));
|
||||
}
|
||||
} else {
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
const auto UseMSBIndex = (Control & 0b0100'0000) != 0;
|
||||
|
||||
OrderedNode *ResultNoFlags = _Bfe(16, 0, IntermediateResult);
|
||||
|
||||
OrderedNode *IfZero = _Constant(16 >> (Control & 1));
|
||||
OrderedNode *IfNotZero = UseMSBIndex ? _FindMSB(ResultNoFlags)
|
||||
: _FindLSB(ResultNoFlags);
|
||||
|
||||
OrderedNode *Result = _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RCX, Result, 4);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags.
|
||||
// We use the top 16-bits of the result to store the flags
|
||||
// in the form:
|
||||
//
|
||||
// Bit: 19 18 17 16
|
||||
// [OF | CF | SF | ZF]
|
||||
//
|
||||
const auto GetFlagBit = [this, IntermediateResult](int BitIndex) {
|
||||
return _Bfe(1, BitIndex, IntermediateResult);
|
||||
};
|
||||
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(GetFlagBit(16));
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(GetFlagBit(17));
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(GetFlagBit(18));
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(GetFlagBit(19));
|
||||
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(ZeroConst);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCMPESTRIOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true, false);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPESTRMOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true, true);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPISTRIOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false, false);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPISTRMOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false, true);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -6,10 +6,10 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
@@ -1374,8 +1374,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
|
||||
SrcCond = _Sbfe(1, 0, SrcCond);
|
||||
|
||||
OrderedNode *VecCond = _VCastFromGPR(16, 8, SrcCond);
|
||||
VecCond = _VInsGPR(16, 8, 1, VecCond, SrcCond);
|
||||
OrderedNode *VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
|
||||
@@ -6,10 +6,10 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
|
||||
+1
-56
@@ -12,56 +12,6 @@ namespace FEXCore {
|
||||
|
||||
thread_local ThreadState ThreadData{};
|
||||
|
||||
static bool IsSynchronous(int Signal) {
|
||||
switch (Signal) {
|
||||
case SIGBUS:
|
||||
case SIGFPE:
|
||||
case SIGILL:
|
||||
case SIGSEGV:
|
||||
case SIGTRAP:
|
||||
return true;
|
||||
default: break;
|
||||
};
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Masks signals from the signal mask
|
||||
*
|
||||
* @param how Argument to sigmask. SIG_{BLOCK, SETMASK, UNBLOCK}
|
||||
* @param Signal Which signal to set or -1 to sweep through them all
|
||||
*/
|
||||
static void MaskSignals(int how, int Signal = -1) {
|
||||
// If we have a helper thread, we need to mask a significant amount of signals so the an errant thread doesn't receive a signal that it shouldn't
|
||||
sigset_t SignalSet{};
|
||||
sigemptyset(&SignalSet);
|
||||
|
||||
if (Signal == -1) {
|
||||
for (int i = 0; i <= SignalDelegator::MAX_SIGNALS; ++i) {
|
||||
// If it is a synchronous signal then don't ignore it
|
||||
if (IsSynchronous(i)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Add this signal to the ignore list
|
||||
sigaddset(&SignalSet, i);
|
||||
}
|
||||
}
|
||||
else {
|
||||
sigaddset(&SignalSet, Signal);
|
||||
}
|
||||
|
||||
// Be warned, a thread will inherit the signal mask if created from this thread
|
||||
int Result = pthread_sigmask(how, &SignalSet, nullptr);
|
||||
if (Result != 0) {
|
||||
LogMan::Msg::EFmt("Couldn't register thread to mask signals");
|
||||
}
|
||||
}
|
||||
|
||||
void SignalDelegator::MaskThreadSignals() {
|
||||
MaskSignals(SIG_BLOCK);
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState *SignalDelegator::GetTLSThread() {
|
||||
return ThreadData.Thread;
|
||||
}
|
||||
@@ -81,18 +31,13 @@ namespace FEXCore {
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetFrontendHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
|
||||
// Let the host take first stab at handling the signal
|
||||
auto Thread = GetTLSThread();
|
||||
HostSignalHandler &Handler = HostHandlers[Signal];
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::EFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
for (auto &Handler : Handler.Handlers) {
|
||||
|
||||
+3
-2
@@ -1,8 +1,9 @@
|
||||
#ifndef NDEBUG
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <tuple>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::X86Tables::X86InstDebugInfo {
|
||||
void InstallDebugInfo() {
|
||||
|
||||
+14
-4
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
@@ -13,13 +14,15 @@ $end_info$
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore {
|
||||
constexpr size_t CODE_SIZE = 0x1000;
|
||||
|
||||
X86GeneratedCode::X86GeneratedCode() {
|
||||
#ifdef _WIN32
|
||||
// No need to allocate anything in this config.
|
||||
#else
|
||||
|
||||
// Allocate a page for our emulated guest
|
||||
CodePtr = AllocateGuestCodeSpace(CODE_SIZE);
|
||||
|
||||
@@ -55,18 +58,22 @@ X86GeneratedCode::X86GeneratedCode() {
|
||||
memcpy(reinterpret_cast<void*>(rt_sigreturn_32), &rt_sigreturn_32_code.at(0), rt_sigreturn_32_code.size());
|
||||
|
||||
mprotect(CodePtr, CODE_SIZE, PROT_READ);
|
||||
#endif
|
||||
}
|
||||
|
||||
X86GeneratedCode::~X86GeneratedCode() {
|
||||
FEXCore::Allocator::munmap(CodePtr, CODE_SIZE);
|
||||
#ifndef _WIN32
|
||||
FEXCore::Allocator::VirtualFree(CodePtr, CODE_SIZE);
|
||||
#endif
|
||||
}
|
||||
|
||||
void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
#ifndef _WIN32
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
if (Is64BitMode()) {
|
||||
// 64bit mode can have its sigret handler anywhere
|
||||
return FEXCore::Allocator::mmap(nullptr, Size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
return FEXCore::Allocator::VirtualAlloc(Size);
|
||||
}
|
||||
|
||||
// First 64bit page
|
||||
@@ -95,6 +102,9 @@ void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
// Can't do anything about this
|
||||
// Here's hoping the application doesn't use signals
|
||||
return MAP_FAILED;
|
||||
#else
|
||||
return nullptr;
|
||||
#endif
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+2
-1
@@ -5,8 +5,9 @@ tags: frontend|x86-tables
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
|
||||
@@ -7,7 +7,6 @@ $end_info$
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -170,10 +169,9 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_DEBUG , 1, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{0xD6, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1, nullptr}},
|
||||
@@ -255,6 +253,9 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
// `L1OM` Larrabee instructions used this as an escape byte.
|
||||
// FEX will never support this.
|
||||
{0xD6, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
@@ -285,6 +286,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD6, 1, X86InstInfo{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
|
||||
@@ -6,8 +6,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
@@ -6,8 +6,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
@@ -5,7 +5,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -6,7 +6,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
#include <iterator>
|
||||
@@ -44,9 +43,9 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
@@ -7,7 +7,6 @@ $end_info$
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
|
||||
|
||||
@@ -6,8 +6,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
@@ -6,8 +6,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
@@ -7,7 +7,6 @@ $end_info$
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
#include <iterator>
|
||||
|
||||
@@ -23,7 +22,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x02, 1, X86InstInfo{"LAR", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x07, 1, X86InstInfo{"SYSRET", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
Loaded 100 of 521 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user