mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 22:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ee56a2cbde | ||
|
|
d66a83a98f | ||
|
|
067b346e1b | ||
|
|
ea7d1697b1 | ||
|
|
5fd91d34f1 | ||
|
|
11880459a5 | ||
|
|
0ef0bb2c97 | ||
|
|
72edee7c6f | ||
|
|
cb5644cf81 | ||
|
|
2dd922c3ce | ||
|
|
b7984e8651 | ||
|
|
31e976a5bc | ||
|
|
009ae55ff0 | ||
|
|
98572b9e23 | ||
|
|
1ab234f6dd | ||
|
|
fed5e6d546 | ||
|
|
f27e2246e2 | ||
|
|
4779fb74de | ||
|
|
eaf83aa6b4 | ||
|
|
4f8b28e83b | ||
|
|
c318947695 | ||
|
|
b3489d7262 | ||
|
|
44d738fa93 | ||
|
|
811487ad98 | ||
|
|
4f4e38ace2 | ||
|
|
edd6becc56 | ||
|
|
e47a94cae7 | ||
|
|
cc82dba1ca | ||
|
|
8232669b22 | ||
|
|
28936073c4 | ||
|
|
ef2559d911 | ||
|
|
d24446ed13 | ||
|
|
67f13ba927 | ||
|
|
dc5239c003 | ||
|
|
f346f89678 | ||
|
|
2f9449cb5a | ||
|
|
139367d248 | ||
|
|
fcebad51bd | ||
|
|
b50292493a | ||
|
|
b74de53056 | ||
|
|
49e798ab2b | ||
|
|
151e2279af | ||
|
|
93ada89708 | ||
|
|
f41674bb7d | ||
|
|
854fd70735 | ||
|
|
78a362581d | ||
|
|
8c0d5c6583 | ||
|
|
1c184997e7 | ||
|
|
ccc699444d | ||
|
|
946c805d84 | ||
|
|
118b8b200e | ||
|
|
aa9d7c5629 | ||
|
|
0b34035085 | ||
|
|
b6bd826014 | ||
|
|
12cc980603 | ||
|
|
f3d55dd721 | ||
|
|
91cef6b76f | ||
|
|
270cbf39b5 | ||
|
|
2e0be0a5e7 | ||
|
|
d60c089697 | ||
|
|
e76ebeab58 | ||
|
|
333271d490 | ||
|
|
a750870abf | ||
|
|
15db72ef60 | ||
|
|
9687ac51f0 | ||
|
|
5f16f357af | ||
|
|
4f028b8614 | ||
|
|
32a4abbea7 | ||
|
|
c00c9b397e | ||
|
|
6b5d8bd8c0 | ||
|
|
7deb4976a3 | ||
|
|
2cfd71c159 | ||
|
|
0f26780de0 | ||
|
|
1e153e0c81 | ||
|
|
6994fc3a01 | ||
|
|
0ef72bf118 | ||
|
|
49ca0e2181 | ||
|
|
947ae1c243 | ||
|
|
d703f3ccee | ||
|
|
d8a18687e8 | ||
|
|
045549f166 | ||
|
|
80e632db8a | ||
|
|
852e3c4e93 | ||
|
|
e86547bbcb | ||
|
|
883cca2e8f | ||
|
|
8540332520 | ||
|
|
df5bdefb8a | ||
|
|
3d1fb7701c | ||
|
|
140976d322 | ||
|
|
9a11d3b1a2 | ||
|
|
96e652879f | ||
|
|
cc1c1dd047 | ||
|
|
c1d572951f | ||
|
|
dd9d3264dd | ||
|
|
2aaf957ad8 | ||
|
|
d459b2f9b5 | ||
|
|
22ab7f2b3e | ||
|
|
f7e32373ce | ||
|
|
25d422e92b | ||
|
|
cfbeece09f | ||
|
|
a597a09825 | ||
|
|
99854ff310 | ||
|
|
44d1502b5a | ||
|
|
c3c635e36e | ||
|
|
7e2f20cabb | ||
|
|
3425b07711 | ||
|
|
9b93495d45 | ||
|
|
cf067994f3 | ||
|
|
0a64f8a9c5 | ||
|
|
3ac7fe3f05 | ||
|
|
be96cb7bd0 | ||
|
|
59ec88f48d | ||
|
|
6ec628fa31 | ||
|
|
90256730c3 | ||
|
|
3a0f9db512 | ||
|
|
5378ae2e76 | ||
|
|
cc9c80d79f | ||
|
|
60e8da05cd | ||
|
|
aac7fa9b58 | ||
|
|
bd4a81a2a1 | ||
|
|
a6211f29e7 | ||
|
|
da263834f8 | ||
|
|
3d671cba10 | ||
|
|
d4be2dc636 | ||
|
|
3e5694bd06 | ||
|
|
66feea9e8e | ||
|
|
2bcd285851 | ||
|
|
30ff225e80 | ||
|
|
8762bc1fa3 | ||
|
|
5b4162b712 | ||
|
|
cb5c07f4b1 | ||
|
|
3364b48f3b | ||
|
|
67530171e6 | ||
|
|
6f00611892 | ||
|
|
67941f04eb | ||
|
|
42531108b7 | ||
|
|
7edab7ee3b | ||
|
|
b9f2389d74 | ||
|
|
3c0b041c44 | ||
|
|
80fcd640be | ||
|
|
0908968e87 | ||
|
|
8e0543d9af | ||
|
|
b902b8edab | ||
|
|
9c38332e7e | ||
|
|
0503c89ff6 | ||
|
|
6dd410698a | ||
|
|
bbac014d6d | ||
|
|
a723ff09c1 | ||
|
|
5769ffbba7 | ||
|
|
a7c7fe4a35 | ||
|
|
80f20ad121 | ||
|
|
808ced455d | ||
|
|
0505b30d34 | ||
|
|
16a5d1a6b1 | ||
|
|
9cab746aa7 | ||
|
|
b888bb5ce5 | ||
|
|
47218254a1 | ||
|
|
7cb15582c2 | ||
|
|
930d2650f8 | ||
|
|
fba5678476 | ||
|
|
68232366e4 | ||
|
|
d7ff1b78fb | ||
|
|
780b48620b | ||
|
|
4a0878fa92 | ||
|
|
67143ed1d1 | ||
|
|
df3d6938ae | ||
|
|
577372c203 | ||
|
|
d4aa64ebd1 | ||
|
|
ba41da7da0 | ||
|
|
806e5b8bcf | ||
|
|
2480bab409 | ||
|
|
de0b690672 | ||
|
|
300e2729a7 | ||
|
|
ad7202e7d7 | ||
|
|
69466dce92 | ||
|
|
175a57dd27 | ||
|
|
e2ce60148c | ||
|
|
99660129f3 | ||
|
|
308d9a751c | ||
|
|
4bd28c0ed8 | ||
|
|
8397f3ac99 | ||
|
|
0452bc7212 | ||
|
|
3d7ed89ffb | ||
|
|
23ab0a978e | ||
|
|
7f47a9ef0e | ||
|
|
4331753ca0 | ||
|
|
cdcc432399 | ||
|
|
fa8bcfd67a | ||
|
|
5e5984a29b | ||
|
|
435b67aa0c | ||
|
|
a1343e9296 | ||
|
|
2dfb7727d5 | ||
|
|
235f32ce8c | ||
|
|
560565aa13 | ||
|
|
2e0cb2fbd4 | ||
|
|
4790a7ba79 | ||
|
|
a4a1d607d5 | ||
|
|
df3e51fc8c | ||
|
|
06c29eab88 | ||
|
|
0139498072 | ||
|
|
8cfbabde94 | ||
|
|
472a701e2b | ||
|
|
a41ebe2d85 | ||
|
|
cce6011205 | ||
|
|
c437129ed8 | ||
|
|
557cb593e1 | ||
|
|
afae24d870 | ||
|
|
8d3f0b6f02 | ||
|
|
60f7b9bcc4 | ||
|
|
a487557173 | ||
|
|
394b4888bb | ||
|
|
142cbdd852 | ||
|
|
2f9102f78d | ||
|
|
c9824d04cb | ||
|
|
9c2a569539 | ||
|
|
515aa4ce3e | ||
|
|
f616beb992 | ||
|
|
0dcf1e12b8 | ||
|
|
2cbf544ef5 | ||
|
|
0eed73beeb | ||
|
|
6993f4fd8d | ||
|
|
920a8db369 | ||
|
|
9c37c0f1c3 | ||
|
|
4623544f69 | ||
|
|
45587278c9 | ||
|
|
ccf1402fe6 | ||
|
|
da0e1b515a | ||
|
|
c0ec4da849 | ||
|
|
bd1e029f35 | ||
|
|
690cb6fa48 | ||
|
|
d34302a968 | ||
|
|
ac77376441 | ||
|
|
f246e9864c | ||
|
|
78cb2cd9fc | ||
|
|
cec1814a09 | ||
|
|
4d49ac7c3d | ||
|
|
6d13d9fb56 | ||
|
|
ae7dc250db | ||
|
|
f4086b25e6 | ||
|
|
8e1aaa0559 | ||
|
|
a4c1aaa6bc | ||
|
|
37a611bcd6 | ||
|
|
e4560ed0c8 | ||
|
|
6d58ea31b9 | ||
|
|
3db31a655c | ||
|
|
e9a6bfc077 | ||
|
|
5a143c016c | ||
|
|
fd1307eea3 | ||
|
|
d9f08d4bdf | ||
|
|
f3eee8f305 | ||
|
|
f66085f4a7 | ||
|
|
c9461d9997 | ||
|
|
d5eb99fac8 | ||
|
|
f3175848b1 | ||
|
|
3dd597a591 | ||
|
|
e8e05252f0 | ||
|
|
93cef53ec0 | ||
|
|
3a19133267 | ||
|
|
9b309b2102 | ||
|
|
fe88b904c9 | ||
|
|
2e63c6d547 | ||
|
|
4cc59ba8f5 | ||
|
|
0bc9e1a409 | ||
|
|
338f12845d | ||
|
|
b3ae81f75f | ||
|
|
c1a1c37980 | ||
|
|
56dd9f23d1 | ||
|
|
fb6f850bb4 | ||
|
|
b6d8749525 | ||
|
|
d3f1397325 | ||
|
|
0a164428fa | ||
|
|
7496175100 | ||
|
|
0616a9cef1 | ||
|
|
97f8775354 | ||
|
|
c92099aa98 | ||
|
|
7c288b09f1 | ||
|
|
680af7b1b0 | ||
|
|
349bc9efab | ||
|
|
ad5c3cb268 | ||
|
|
be8d37ef3d | ||
|
|
6ad2514bfe | ||
|
|
3fa6129a14 | ||
|
|
a57cebaf58 | ||
|
|
34fdb14da1 | ||
|
|
974baca09c | ||
|
|
f22094a493 | ||
|
|
d979b3a1da | ||
|
|
6d82c957fa | ||
|
|
28715a1df6 | ||
|
|
b40d8f5679 | ||
|
|
2884337d85 | ||
|
|
3036f3b3ff | ||
|
|
b02ed40b5a | ||
|
|
0114c874a7 | ||
|
|
3ff3dc8769 | ||
|
|
d04c94fe80 | ||
|
|
0ebf260ed0 | ||
|
|
28ae84cf60 | ||
|
|
26007168b0 | ||
|
|
a48237e218 | ||
|
|
fa3352004e | ||
|
|
b9378856da | ||
|
|
515e98f4cb | ||
|
|
ce2924731e | ||
|
|
c099fd01f1 | ||
|
|
36250b10f6 | ||
|
|
bc67910ee4 | ||
|
|
79526b9c9e | ||
|
|
31a4158957 | ||
|
|
58f3d3caf5 | ||
|
|
be52676b3b | ||
|
|
ae48228943 | ||
|
|
e8e35e48c7 | ||
|
|
8b8f27a88f | ||
|
|
8e7906a665 | ||
|
|
85105a6af3 | ||
|
|
027fbbf051 | ||
|
|
ca31a0404c | ||
|
|
f644959c7c | ||
|
|
10cca02656 | ||
|
|
16a54742e6 | ||
|
|
ad9aa0bc87 | ||
|
|
bd2b3f35a3 | ||
|
|
50169ce640 | ||
|
|
baae2d68f9 | ||
|
|
ad8d038b8a | ||
|
|
820932e3c7 | ||
|
|
6f11f2e6f4 | ||
|
|
409b6ff6ef | ||
|
|
f5ad7682c3 | ||
|
|
04805f351b | ||
|
|
8e3d4a3e02 | ||
|
|
0913741343 | ||
|
|
929193c16c | ||
|
|
69ff984a82 | ||
|
|
c5ffc0664d | ||
|
|
c1ef11f034 | ||
|
|
750b0b70bc | ||
|
|
56d8080ec9 | ||
|
|
c0be974272 | ||
|
|
0e97f8fb2b | ||
|
|
c4c10d0a16 | ||
|
|
576cf61a96 | ||
|
|
e323938173 | ||
|
|
407e26bfee | ||
|
|
d04f3a288e | ||
|
|
3bfb4bedd5 | ||
|
|
fb15cf0db4 | ||
|
|
b08d372c78 | ||
|
|
acf48e29b2 | ||
|
|
162733130c | ||
|
|
f6ff6c8b44 | ||
|
|
6efc4a967c | ||
|
|
a6c57f71e9 | ||
|
|
2af7e997f4 | ||
|
|
ab6c00bbcf | ||
|
|
e18453cb57 | ||
|
|
39f49782da | ||
|
|
2c5dd20f3c | ||
|
|
136fa78825 | ||
|
|
190d8020ff | ||
|
|
6eaeb48fac | ||
|
|
f956f008ea | ||
|
|
be4d1a8860 | ||
|
|
8ff4b525fc | ||
|
|
832072362b | ||
|
|
1f7a619c79 | ||
|
|
ddfd393789 | ||
|
|
d9985eb877 | ||
|
|
58127bd0e8 | ||
|
|
e8945dfb6d | ||
|
|
0a6e875329 | ||
|
|
cd4578de27 | ||
|
|
295fe03eef | ||
|
|
4cbaad298a | ||
|
|
dc477c3bd7 | ||
|
|
52419d9911 | ||
|
|
8c3163096b | ||
|
|
9f311cd97e | ||
|
|
1115ce4a95 | ||
|
|
03b802cf8e | ||
|
|
615cfe0246 | ||
|
|
dae16aaf18 | ||
|
|
3d5f876585 | ||
|
|
37102400b5 | ||
|
|
c01e6283ae | ||
|
|
248dc97993 | ||
|
|
d488592eda | ||
|
|
743df8dfae | ||
|
|
4b3792196f | ||
|
|
c333aac4f9 | ||
|
|
db7d7a6bd7 | ||
|
|
04a88ed3ab | ||
|
|
9da08b40bd | ||
|
|
5467c3e478 | ||
|
|
9841983955 | ||
|
|
4e7bab849c | ||
|
|
d098545c20 | ||
|
|
5358af7794 | ||
|
|
eea2e7bb57 | ||
|
|
25bcddf3a5 | ||
|
|
f785b38e4d | ||
|
|
0071c1bda2 | ||
|
|
058e691ef1 | ||
|
|
d806db53ec | ||
|
|
bcd0efa724 | ||
|
|
48c2e0689a | ||
|
|
4b09a6bee1 | ||
|
|
af645cb750 | ||
|
|
b8b9b2f1a7 | ||
|
|
f02bd5c9a6 | ||
|
|
2b39813581 | ||
|
|
923c53c4ea | ||
|
|
66658d82c6 | ||
|
|
b115c144fb | ||
|
|
d8f20751fe | ||
|
|
1977747fc2 | ||
|
|
257016bf12 | ||
|
|
69d65fba4a | ||
|
|
bce694ebb5 | ||
|
|
5d37d5db1a | ||
|
|
4d109c9ce0 | ||
|
|
db9b326534 | ||
|
|
21ae8336cd | ||
|
|
c6f8901c16 | ||
|
|
a797699d62 | ||
|
|
36c524a021 | ||
|
|
dbf33523bd | ||
|
|
de2cd469a2 | ||
|
|
1c34b25538 | ||
|
|
38ad3f0e05 | ||
|
|
266f7feecb | ||
|
|
f9902142f7 | ||
|
|
175d407c9f | ||
|
|
ae2f98e017 | ||
|
|
9e5d7aa5fe | ||
|
|
8648fb1485 | ||
|
|
8b24f7fc26 | ||
|
|
b1c3737aef | ||
|
|
00669a1c89 | ||
|
|
1cedc3d85a | ||
|
|
5e26b77a7c | ||
|
|
58f2693954 | ||
|
|
b4b8e81f24 | ||
|
|
82ce76b1d3 | ||
|
|
a8f797d36b | ||
|
|
93ec676ce8 | ||
|
|
362cccc210 | ||
|
|
c45dc6dd48 | ||
|
|
3d2cbc5d08 | ||
|
|
81c85d73b2 | ||
|
|
5b4e9c6907 | ||
|
|
aa2e8704bc | ||
|
|
cf86ae6b65 | ||
|
|
68d6cf5f14 | ||
|
|
c7bc15785b | ||
|
|
c3d2d01c1f | ||
|
|
9dda960529 | ||
|
|
f2bcc14cda | ||
|
|
86654907bf | ||
|
|
4333261639 | ||
|
|
12b72f908b | ||
|
|
f5997a084e | ||
|
|
6c8a54ff84 | ||
|
|
bcc2901d7f | ||
|
|
358bbb51ff | ||
|
|
f2da70c7e7 | ||
|
|
1a2f41922c | ||
|
|
0ede707e0b | ||
|
|
c4c31d4e80 | ||
|
|
12923ba1b7 | ||
|
|
bd13052708 | ||
|
|
ec89a00a92 | ||
|
|
e657a27607 | ||
|
|
8bb5462554 | ||
|
|
98f21a2a28 | ||
|
|
0a4e064da4 | ||
|
|
5660065eea | ||
|
|
131bf48f4b | ||
|
|
a1cf14f2d9 | ||
|
|
499a40ecb5 | ||
|
|
7524029a06 | ||
|
|
fe3df001e2 | ||
|
|
2053c712a5 | ||
|
|
5c6f229e76 | ||
|
|
acdb4c7061 | ||
|
|
b613576ac4 | ||
|
|
8f52d8e719 | ||
|
|
0fe5e3d1e7 | ||
|
|
26c9d5d427 | ||
|
|
60b0852cde | ||
|
|
3a5ac39bb1 | ||
|
|
301310329b | ||
|
|
86c6ca3049 | ||
|
|
eb5cf1a667 | ||
|
|
f515b1eeb1 | ||
|
|
924cfb5429 | ||
|
|
2a2c389be6 | ||
|
|
aaef83b777 | ||
|
|
29d82cda48 | ||
|
|
9f721e96db | ||
|
|
281be3aeeb | ||
|
|
5b87307452 | ||
|
|
aebabf6ef2 | ||
|
|
0d5fd5dbaf | ||
|
|
3729b02255 | ||
|
|
d4e9ed8e61 | ||
|
|
c10402f4f9 | ||
|
|
64276dbd0c | ||
|
|
b619f381b3 | ||
|
|
3b0aff5fb9 | ||
|
|
a8ab8bbe8e | ||
|
|
3e2ba6d835 | ||
|
|
c6497fe32b | ||
|
|
c8ef77c15f | ||
|
|
b35fadf7e3 | ||
|
|
250ffb6d23 | ||
|
|
f6b1434d63 | ||
|
|
6bbae69c75 | ||
|
|
d0f54bcb23 | ||
|
|
470615b896 | ||
|
|
b02ab8ee19 | ||
|
|
5f6046be4c | ||
|
|
e923e83efb | ||
|
|
e836e4212d | ||
|
|
9417c93110 | ||
|
|
068599b1ec | ||
|
|
7216415bfc | ||
|
|
3020626506 | ||
|
|
0a79fa8d5d | ||
|
|
f8380b9adb | ||
|
|
d898028bc3 | ||
|
|
f090700184 | ||
|
|
01d29dffb9 | ||
|
|
14ba64a22d | ||
|
|
6716077cb6 | ||
|
|
8892580c41 | ||
|
|
1b41304fc1 | ||
|
|
afebf73e8f | ||
|
|
7de66ac3a4 | ||
|
|
85a1c1ff25 | ||
|
|
8e892ece59 | ||
|
|
aa1344aadd | ||
|
|
3f02d7c665 | ||
|
|
f328fca880 | ||
|
|
47d79978ef | ||
|
|
2e24f34a3f | ||
|
|
6e8af295c5 | ||
|
|
bba156a3c1 | ||
|
|
8015ce2099 | ||
|
|
1153c1a538 | ||
|
|
a47b3cccb8 | ||
|
|
2070056d16 | ||
|
|
e227f1343f | ||
|
|
3c7335713d | ||
|
|
bdf4089264 | ||
|
|
b027113998 | ||
|
|
389c6b11dd | ||
|
|
fa5d9dc3b7 | ||
|
|
cb56728e57 | ||
|
|
b89c3a4573 | ||
|
|
0cc11108ba | ||
|
|
a7caf83022 | ||
|
|
053452c40c | ||
|
|
d4361c87ae | ||
|
|
6469eb7a0e | ||
|
|
13fbd0e802 | ||
|
|
e555a8f817 | ||
|
|
f9fb61cf1a | ||
|
|
43cf2e4e2c | ||
|
|
98f9a65202 | ||
|
|
8726c8fb73 | ||
|
|
27f3cb336f | ||
|
|
aa3bacd938 | ||
|
|
c71492ef32 | ||
|
|
70191f2d28 | ||
|
|
93db8b7ca7 | ||
|
|
d33b0cb9e3 | ||
|
|
0806d4ec25 | ||
|
|
9b646746b2 | ||
|
|
11993daec4 | ||
|
|
a78ffeeaba | ||
|
|
05b78339f6 | ||
|
|
365c221029 | ||
|
|
f60608a9c0 | ||
|
|
153d871be2 | ||
|
|
b69f2d7773 | ||
|
|
09ffe7ef6b | ||
|
|
3dfb94b524 | ||
|
|
d1e43d94e9 | ||
|
|
1b490e0e53 | ||
|
|
094146d630 | ||
|
|
23c2a53683 | ||
|
|
82b7689ca4 | ||
|
|
149f3e6f6d | ||
|
|
cea551c2ac | ||
|
|
2dcae23776 | ||
|
|
92e4e75217 | ||
|
|
5ca35bf77c | ||
|
|
c956b82d27 | ||
|
|
1c115096c4 | ||
|
|
c1d5fae018 | ||
|
|
17d49fc00f | ||
|
|
0e1e4c16b1 | ||
|
|
bec8e27b4f | ||
|
|
56841f0e50 | ||
|
|
723146050b | ||
|
|
85b1aa4c2d | ||
|
|
0506369519 | ||
|
|
ba1632974e | ||
|
|
c69082b1a4 | ||
|
|
74b2548982 | ||
|
|
f31656ec65 | ||
|
|
e91420c405 | ||
|
|
25df59a65d | ||
|
|
83fdd5720f | ||
|
|
db63241fd4 | ||
|
|
89b00c89aa | ||
|
|
d38917b5f0 | ||
|
|
910e0242c1 | ||
|
|
651b7bb75d | ||
|
|
c9f13ae1dd | ||
|
|
862e575100 | ||
|
|
4669c4541c | ||
|
|
769a8c41c4 | ||
|
|
bec9dba2b1 | ||
|
|
205ba2ea13 | ||
|
|
2073f6d287 | ||
|
|
0f25a960ee | ||
|
|
282ed3e309 | ||
|
|
cd031a7d38 | ||
|
|
e1885ed0bd | ||
|
|
4a31b619fa | ||
|
|
bd4464bd5e | ||
|
|
d907a7dc9f | ||
|
|
732070f750 | ||
|
|
48442b6b03 | ||
|
|
7fdbe547a3 | ||
|
|
59565b828d | ||
|
|
1e2d059890 | ||
|
|
0aa41908a2 | ||
|
|
b27ce3f79c | ||
|
|
238e52f74a | ||
|
|
9398b931fb | ||
|
|
b2a9785959 | ||
|
|
224a1f19a3 | ||
|
|
109c53f22b | ||
|
|
e25849b2cb | ||
|
|
57978accc1 | ||
|
|
ff37177f4d | ||
|
|
472d143021 | ||
|
|
157f95b08f | ||
|
|
6bf7ab0778 | ||
|
|
0de958be2a | ||
|
|
ef544fecf2 | ||
|
|
5eea68d6c6 | ||
|
|
5471367db1 | ||
|
|
b8265b1067 | ||
|
|
099c683a5a | ||
|
|
0357bb23e8 | ||
|
|
c7193b52fb | ||
|
|
ec14a65e23 | ||
|
|
0d70c6a0d0 | ||
|
|
bfa069c4d5 | ||
|
|
61bdf64e15 | ||
|
|
482b35c283 | ||
|
|
b74d886017 | ||
|
|
1667abad7e | ||
|
|
b187a853e7 | ||
|
|
228c7d142e | ||
|
|
041199644c | ||
|
|
3767f3633d | ||
|
|
af3253947e | ||
|
|
efc5eb2933 | ||
|
|
b4eeb96375 | ||
|
|
c9832e3d34 | ||
|
|
da3e3fc7a3 | ||
|
|
b5c83f0628 | ||
|
|
03087a55ba | ||
|
|
1ce3c16b30 | ||
|
|
bf702850a9 | ||
|
|
584c4cc05e | ||
|
|
279afd88bb | ||
|
|
c0a6d82025 | ||
|
|
3a03e1c93c | ||
|
|
bdaa70405f | ||
|
|
1281145982 | ||
|
|
87cac09477 | ||
|
|
5336129b58 | ||
|
|
72fc2b522d | ||
|
|
b6f6c84790 | ||
|
|
11e9be13b1 | ||
|
|
afdb8753ba | ||
|
|
f6a2e6739d | ||
|
|
d6569d510d | ||
|
|
04e4993d9b | ||
|
|
783e09d67d | ||
|
|
314f478225 | ||
|
|
c1dbc28aa2 | ||
|
|
8f7e393ffb | ||
|
|
b3055523b4 | ||
|
|
cf6b21564c | ||
|
|
996a4c023c | ||
|
|
0dcbdcc0e2 | ||
|
|
5bdd422db6 | ||
|
|
bf147f47b5 | ||
|
|
3f1f7faf34 | ||
|
|
1fc6725826 | ||
|
|
fa8c35feba | ||
|
|
73958b9163 | ||
|
|
9b81a83894 | ||
|
|
65b7d4007e | ||
|
|
2c0444e846 | ||
|
|
b92e716d0c | ||
|
|
a7a1365cf7 | ||
|
|
8ee5b5cf50 | ||
|
|
b45023bedf | ||
|
|
3702e513f5 | ||
|
|
829384e488 | ||
|
|
ed23fbe932 | ||
|
|
0af0427efd | ||
|
|
a499272d81 | ||
|
|
5dee921300 | ||
|
|
c0dcf8925a | ||
|
|
e91c5ff906 | ||
|
|
190f7c27e0 | ||
|
|
b15f0b5d36 | ||
|
|
e2c65189ff | ||
|
|
03f63f99a8 | ||
|
|
e305a9a0d5 | ||
|
|
3a90dbbb35 | ||
|
|
5103f2d92b | ||
|
|
d4a6b031ea | ||
|
|
319cf4bf3d | ||
|
|
d75c0f2c50 | ||
|
|
a586d3823d | ||
|
|
5522c6db9c | ||
|
|
367e1658ad | ||
|
|
bbad06f81a | ||
|
|
9612b2fe4b | ||
|
|
ef5503f0b7 | ||
|
|
26bf67ca76 | ||
|
|
e5df636efd | ||
|
|
f34f4a0227 | ||
|
|
f45722d2cd | ||
|
|
db7ef0e4b0 | ||
|
|
de10cbad98 | ||
|
|
e4d9c264d8 | ||
|
|
97b3efa90a | ||
|
|
18065199e3 | ||
|
|
77d92872bc | ||
|
|
460f13be71 | ||
|
|
47c9463217 | ||
|
|
15c825f362 | ||
|
|
bbd20b47ba | ||
|
|
5b70209728 | ||
|
|
a287f2a189 | ||
|
|
3a240e3b61 | ||
|
|
de1e593ec2 | ||
|
|
13cd8b33a2 | ||
|
|
0f26bc20a3 | ||
|
|
eacab3cc22 | ||
|
|
09e3371a0d | ||
|
|
ff3f7345b6 | ||
|
|
8181e53727 | ||
|
|
5431aa5a28 | ||
|
|
1a293cc542 | ||
|
|
b1e78934ad | ||
|
|
61f22911c7 | ||
|
|
fe8778bb96 | ||
|
|
14e5ea1e22 | ||
|
|
74f1205f33 | ||
|
|
4045bfd187 | ||
|
|
9db93a43dd | ||
|
|
a379d50729 | ||
|
|
f5822f83b0 | ||
|
|
5028434292 | ||
|
|
8f0461cac8 | ||
|
|
0b4cb23411 | ||
|
|
bab96b9441 | ||
|
|
11e2f14185 | ||
|
|
6177290e9d | ||
|
|
20a54913bd | ||
|
|
dd9ed89a7a | ||
|
|
a4e1e0a1fb | ||
|
|
5e9f69001d | ||
|
|
39c5ab1c81 | ||
|
|
5bf790324c | ||
|
|
adcdb32d49 | ||
|
|
f264578f12 | ||
|
|
7149da387a | ||
|
|
0f3d14e7c0 | ||
|
|
aad5080224 | ||
|
|
c77a3d673c | ||
|
|
0ff2e6e1e3 | ||
|
|
8538f5bac4 | ||
|
|
a305baf6e5 | ||
|
|
807619aa02 | ||
|
|
6db2125b41 | ||
|
|
9f6d80fe5d | ||
|
|
423ce12001 | ||
|
|
4edd72fc33 | ||
|
|
978f607dd9 | ||
|
|
9ba78c9771 | ||
|
|
e2144345c0 | ||
|
|
63e4c3682d | ||
|
|
2956e84ead | ||
|
|
4466c50c2b | ||
|
|
dd5ca1d349 | ||
|
|
95c756b466 | ||
|
|
99465faf63 | ||
|
|
e018917f76 | ||
|
|
a261d9909e | ||
|
|
6de8bc6848 | ||
|
|
d87155e4ee | ||
|
|
7484cacaf9 | ||
|
|
42259974c4 | ||
|
|
e455996dbd | ||
|
|
b5dd1d05e9 | ||
|
|
bbaf70da15 | ||
|
|
65e8d094ef | ||
|
|
4c801d594a | ||
|
|
d4403edea9 | ||
|
|
06ef012fb2 | ||
|
|
8f8f37684a | ||
|
|
7140b8d901 | ||
|
|
d5beba9423 | ||
|
|
826e15aea9 | ||
|
|
14e80ce228 | ||
|
|
165d3d3d4d | ||
|
|
887200e571 | ||
|
|
2c0bc0654d | ||
|
|
b3d76bd2f1 | ||
|
|
2e694412f4 | ||
|
|
d84577c36c | ||
|
|
cf9c2aa72c | ||
|
|
1cb8e4891c | ||
|
|
24f2796141 | ||
|
|
cb215b5f21 | ||
|
|
0cf2695772 | ||
|
|
6a6886305e | ||
|
|
5ef7537e61 | ||
|
|
167fe85cc3 | ||
|
|
cf65747667 | ||
|
|
27bb28b47f | ||
|
|
a00da800e7 | ||
|
|
bf835e80ac | ||
|
|
8f246b206b | ||
|
|
3c5c23bf36 | ||
|
|
5bcfaf4b9f | ||
|
|
1f6c6345d9 | ||
|
|
93792577eb | ||
|
|
3d23cd5765 | ||
|
|
fcc239552c | ||
|
|
8238de024f | ||
|
|
39e658f02a | ||
|
|
e89dd27f2a | ||
|
|
f85fae0041 | ||
|
|
65eec673fc | ||
|
|
5c93a085d2 | ||
|
|
4b356a7c2c | ||
|
|
2b67f87054 | ||
|
|
1ea40ae676 | ||
|
|
e0ef32e0bf | ||
|
|
a2b53c8eb0 | ||
|
|
21b6cccb4e | ||
|
|
d539829251 | ||
|
|
ef321e4bf8 | ||
|
|
47a0f14537 | ||
|
|
6d39f369b0 | ||
|
|
2304cfc530 | ||
|
|
1a39de4509 | ||
|
|
efb479f88f | ||
|
|
b27bf43901 | ||
|
|
c612fa8f2f | ||
|
|
4ccc40f697 | ||
|
|
cb53a704ba | ||
|
|
483423674a | ||
|
|
180d16af7a | ||
|
|
1acc038826 | ||
|
|
cc558fd5dc | ||
|
|
8f04223193 | ||
|
|
4be649c44e | ||
|
|
6253f4f708 | ||
|
|
cc2eef619c | ||
|
|
3bff42e6a7 | ||
|
|
2671246fef | ||
|
|
8cb8f090dd | ||
|
|
a37d89a7d5 | ||
|
|
252d7712ea | ||
|
|
c548625fbe | ||
|
|
f036a0b84f | ||
|
|
2e1389b25e | ||
|
|
a5f82a57fa | ||
|
|
cd83d3eb24 | ||
|
|
93ab8ab23c | ||
|
|
462fff2c67 | ||
|
|
8dab35cbf8 | ||
|
|
a1a479e69f | ||
|
|
6403290019 | ||
|
|
580bd50a00 | ||
|
|
b2a8b0ca12 | ||
|
|
f78bdf0852 | ||
|
|
5652eb4c5d | ||
|
|
a52bb47551 | ||
|
|
22590dde77 | ||
|
|
6543a80ff9 | ||
|
|
9c36d1061b | ||
|
|
5a3cc7b469 | ||
|
|
4cff3e5f1f | ||
|
|
a1eb571630 | ||
|
|
559cf6491a | ||
|
|
4bdda1eeb5 | ||
|
|
fc70fc3506 | ||
|
|
26ee63cc24 | ||
|
|
0092ea7c0b | ||
|
|
439a3b9c3a | ||
|
|
b4ddf36582 | ||
|
|
12c44f26e5 | ||
|
|
5b7ba06d5c |
No files matched your search
@@ -37,7 +37,6 @@ If applicable, add screenshots and video to help explain your problem.
|
||||
|
||||
**Additional context**
|
||||
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
|
||||
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
|
||||
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
|
||||
- Is this a Vulkan game: [Yes/No/Unknown]
|
||||
- If Yes, What is your Vulkan driver:
|
||||
|
||||
@@ -13,7 +13,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
@@ -78,18 +77,6 @@ jobs:
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
|
||||
@@ -20,7 +20,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
@@ -85,18 +84,6 @@ jobs:
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
|
||||
@@ -56,6 +56,16 @@ jobs:
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Set vixl_sim x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
|
||||
|
||||
- name: Set vixl_sim Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
@@ -64,7 +74,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=False -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -100,18 +100,6 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM128bit.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
+6
-6
@@ -15,19 +15,19 @@
|
||||
path = External/tiny-json
|
||||
url = https://github.com/Sonicadvance1/tiny-json.git
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
url = https://github.com/FEX-Emu/xbyak.git
|
||||
url = https://github.com/herumi/xbyak.git
|
||||
[submodule "External/fex-posixtest-bins"]
|
||||
shallow = true
|
||||
shallow = true
|
||||
path = External/fex-posixtest-bins
|
||||
url = https://github.com/FEX-Emu/fex-posixtest-bins.git
|
||||
[submodule "External/fex-gvisor-tests-bins"]
|
||||
shallow = true
|
||||
shallow = true
|
||||
path = External/fex-gvisor-tests-bins
|
||||
url = https://github.com/FEX-Emu/fex-gvisor-tests-bins.git
|
||||
[submodule "External/fex-gcc-target-tests-bins"]
|
||||
shallow = true
|
||||
shallow = true
|
||||
path = External/fex-gcc-target-tests-bins
|
||||
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
|
||||
[submodule "External/jemalloc"]
|
||||
@@ -41,7 +41,7 @@
|
||||
url = https://github.com/FEX-Emu/drm-headers.git
|
||||
[submodule "External/xxhash"]
|
||||
path = External/xxhash
|
||||
url = https://github.com/FEX-Emu/xxHash.git
|
||||
url = https://github.com/Cyan4973/xxHash.git
|
||||
[submodule "External/Catch2"]
|
||||
path = External/Catch2
|
||||
url = https://github.com/catchorg/Catch2.git
|
||||
|
||||
+15
-7
@@ -122,6 +122,11 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
|
||||
set(_M_ARM_64EC 1)
|
||||
add_definitions(-D_M_ARM_64EC=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_CCACHE)
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
@@ -232,8 +237,10 @@ endif()
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
@@ -293,10 +300,11 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set(FEX_TUNE_COMPILE_FLAGS)
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
@@ -309,7 +317,7 @@ if (TUNE_CPU STREQUAL "native")
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
@@ -323,19 +331,19 @@ if (TUNE_CPU STREQUAL "native")
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${AARCH64_CPU}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
add_compile_options("-march=native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${TUNE_CPU}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
|
||||
endif()
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"AdditionalArguments": "--no-sandbox"
|
||||
}
|
||||
}
|
||||
+5
-4
@@ -3,11 +3,12 @@ FROM ubuntu:20.04 as builder
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
|
||||
clang-10 llvm-10 nasm ninja-build \
|
||||
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
|
||||
python3 linux-headers-generic
|
||||
clang-10 llvm-10 nasm ninja-build pkg-config \
|
||||
libcap-dev libglfw3-dev libepoxy-dev python3-dev libsdl2-dev \
|
||||
python3 linux-headers-generic \
|
||||
git
|
||||
|
||||
COPY . /opt/FEX
|
||||
RUN git clone --recurse-submodules https://github.com/FEX-Emu/FEX.git
|
||||
|
||||
CMD [ "mkdir /opt/FEX/build" ]
|
||||
|
||||
|
||||
Vendored
-13
@@ -1,13 +0,0 @@
|
||||
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
|
||||
Version 2, December 2004
|
||||
|
||||
Copyright (C) 2018 Ryan Houdek <Sonicadvance1@gmail.com>
|
||||
|
||||
Everyone is permitted to copy and distribute verbatim or modified
|
||||
copies of this license document, and changing it is allowed as long
|
||||
as the name is changed.
|
||||
|
||||
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
|
||||
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
|
||||
|
||||
0. You just DO WHAT THE FUCK YOU WANT TO.
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 85c2334e92...31aa7f634b.
Vendored
+1
-1
Submodule External/drm-headers updated: a11dbbb452...07099adb70.
Vendored
+1
-1
Submodule External/fmt updated: e57ca2e368...f5e54359df.
Vendored
+1
-1
Submodule External/vixl updated: debc345683...7725aec177.
Vendored
+1
-1
Submodule External/xbyak updated: 5f8c0488ba...f17cb9d6b9.
Vendored
+1
-1
Submodule External/xxhash updated: ba7375d54f...bbb27a5efb.
@@ -30,15 +30,21 @@ set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
__attribute__((preserve_all))
|
||||
void Testy() {
|
||||
int Testy(int a, int b, int c, int d, int e, int f) {
|
||||
return a + b + c + d + e + f;
|
||||
}
|
||||
int main() {
|
||||
return 0;
|
||||
return Testy(0, 1, 2, 3, 4, 5);
|
||||
}"
|
||||
HAS_CLANG_PRESERVE_ALL)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
if (HAS_CLANG_PRESERVE_ALL)
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
if (MINGW_BUILD)
|
||||
message(STATUS "Ignoring broken clang::preserve_all support")
|
||||
set(HAS_CLANG_PRESERVE_ALL FALSE)
|
||||
else()
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
|
||||
@@ -441,6 +441,24 @@ def print_parse_envloader_options(options):
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_jsonloader_options(options):
|
||||
output_argloader.write("#ifdef JSONLOADER\n")
|
||||
output_argloader.write("#undef JSONLOADER\n")
|
||||
output_argloader.write("if (false) {}\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("else {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_enum_options(options):
|
||||
output_argloader.write("#ifdef ENUMDEFINES\n")
|
||||
output_argloader.write("#undef ENUMDEFINES\n")
|
||||
@@ -556,6 +574,9 @@ print_parse_argloader_options(options);
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
# Generate json loader code
|
||||
print_parse_jsonloader_options(options);
|
||||
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
|
||||
@@ -46,11 +46,15 @@ class OpDefinition:
|
||||
NumElements: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
RAOverride: int
|
||||
SwitchGen: bool
|
||||
ArgPrinter: bool
|
||||
SSAArgNum: int
|
||||
NonSSAArgNum: int
|
||||
DynamicDispatch: bool
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
@@ -64,11 +68,15 @@ class OpDefinition:
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
self.ImplicitFlagClobber = False
|
||||
self.RAOverride = -1
|
||||
self.SwitchGen = True
|
||||
self.ArgPrinter = True
|
||||
self.SSAArgNum = 0
|
||||
self.NonSSAArgNum = 0
|
||||
self.DynamicDispatch = False
|
||||
self.JITDispatch = True
|
||||
self.JITDispatchOverride = None
|
||||
self.Arguments = []
|
||||
self.EmitValidation = []
|
||||
self.Desc = []
|
||||
@@ -213,6 +221,9 @@ def parse_ops(ops):
|
||||
if "HasSideEffects" in op_val:
|
||||
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
|
||||
|
||||
if "ImplicitFlagClobber" in op_val:
|
||||
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
|
||||
|
||||
if "ArgPrinter" in op_val:
|
||||
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
|
||||
|
||||
@@ -228,6 +239,15 @@ def parse_ops(ops):
|
||||
if "Desc" in op_val:
|
||||
OpDef.Desc = op_val["Desc"]
|
||||
|
||||
if "DynamicDispatch" in op_val:
|
||||
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
|
||||
|
||||
if "JITDispatch" in op_val:
|
||||
OpDef.JITDispatch = bool(op_val["JITDispatch"])
|
||||
|
||||
if "JITDispatchOverride" in op_val:
|
||||
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -357,6 +377,7 @@ def print_ir_sizes():
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
@@ -450,15 +471,17 @@ def print_ir_getraargs():
|
||||
def print_ir_hassideeffects():
|
||||
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
|
||||
for array, prop in [("SideEffects", "HasSideEffects"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
|
||||
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
|
||||
|
||||
output_file.write("};\n\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("bool HasSideEffects(IROps Op) {\n")
|
||||
output_file.write(" return SideEffects[Op];\n")
|
||||
output_file.write("}\n")
|
||||
output_file.write(f"bool {prop}(IROps Op) {{\n")
|
||||
output_file.write(f" return {array}[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -627,6 +650,10 @@ def print_ir_allocator_helpers():
|
||||
|
||||
output_file.write(") {\n")
|
||||
|
||||
# Save NZCV if needed before clobbering NZCV
|
||||
if op.ImplicitFlagClobber:
|
||||
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
|
||||
|
||||
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
@@ -675,7 +702,8 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
@@ -730,10 +758,38 @@ def print_ir_parser_switch_helper():
|
||||
output_file.write("#undef IROP_PARSER_SWITCH_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
if (len(sys.argv) < 3):
|
||||
def print_ir_dispatcher_defs():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DEFS\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.SwitchGen and op.JITDispatch and op.JITDispatchOverride == None:
|
||||
output_dispatch_file.write("DEF_OP({});\n".format(op.Name))
|
||||
|
||||
output_dispatch_file.write("#undef IROP_DISPATCH_DEFS\n")
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DISPATCH\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.JITDispatch:
|
||||
DispatchName = op.Name
|
||||
if op.JITDispatchOverride != None:
|
||||
DispatchName = op.JITDispatchOverride
|
||||
|
||||
if (op.DynamicDispatch):
|
||||
output_dispatch_file.write("REGISTER_OP_RT({}, {});\n".format(op.Name.upper(), DispatchName))
|
||||
else:
|
||||
output_dispatch_file.write("REGISTER_OP({}, {});\n".format(op.Name.upper(), DispatchName))
|
||||
|
||||
output_dispatch_file.write("#undef IROP_DISPATCH_DISPATCH\n")
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
|
||||
json_file = open(sys.argv[1], "r")
|
||||
json_text = json_file.read()
|
||||
json_file.close()
|
||||
@@ -763,3 +819,10 @@ print_ir_allocator_helpers()
|
||||
print_ir_parser_switch_helper()
|
||||
|
||||
output_file.close()
|
||||
|
||||
output_dispatch_file = open(output_dispatcher_filename, "w")
|
||||
print_ir_dispatcher_defs()
|
||||
print_ir_dispatcher_dispatch()
|
||||
|
||||
output_dispatch_file.close()
|
||||
|
||||
@@ -7,6 +7,7 @@ set (FEXCORE_BASE_SRCS
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/SpinWaitLock.cpp
|
||||
)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
@@ -90,7 +91,6 @@ set (SRCS
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
@@ -101,9 +101,7 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
@@ -152,7 +150,6 @@ set (SRCS
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/InlineCallOptimization.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
@@ -197,7 +194,7 @@ endif()
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
|
||||
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
@@ -221,15 +218,16 @@ configure_file(
|
||||
# Generate IR include file
|
||||
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(OUTPUT_DISPATCHER_NAME "${OUTPUT_IR_FOLDER}/IRDefines_Dispatch.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
|
||||
@@ -361,6 +359,7 @@ function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
@@ -369,6 +368,7 @@ endfunction()
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
if (MINGW_BUILD)
|
||||
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -8,6 +7,7 @@
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -321,16 +321,6 @@ namespace DefaultValues {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
{
|
||||
// Always fix up the number of threads and create the configuration
|
||||
// Otherwise the application could receive zero as the number of threads
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
@@ -31,15 +31,6 @@
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
@@ -82,7 +73,13 @@
|
||||
"ENABLEFLAGM": "enableflagm",
|
||||
"DISABLEFLAGM": "disableflagm",
|
||||
"ENABLEFLAGM2": "enableflagm2",
|
||||
"DISABLEFLAGM2": "disableflagm2"
|
||||
"DISABLEFLAGM2": "disableflagm2",
|
||||
"ENABLECRYPTO": "enablecrypto",
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
"DISABLERPRES": "disablerpres",
|
||||
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
|
||||
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -100,7 +97,30 @@
|
||||
"\t{enable,disable}atomics: Will force enable or disable ARMv8.1 LSE atomics even if the host doesn't support it",
|
||||
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it",
|
||||
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it"
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
|
||||
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
|
||||
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"CPUID": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::CPUID::OFF",
|
||||
"Enums": {
|
||||
"ENABLESHA": "enablesha",
|
||||
"DISABLESHA": "disablesha"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features are exposed in CPUID.",
|
||||
"\toff: Default CPU features queried from CPU features",
|
||||
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Scales the cycle counter on systems that have low frequencies."
|
||||
]
|
||||
}
|
||||
},
|
||||
@@ -250,23 +270,6 @@
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
},
|
||||
"SRA": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Set to false to disable Static Register Allocation"
|
||||
]
|
||||
},
|
||||
"Force32BitAllocator": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Forces use of the 32-bit allocator on 32-bit applications",
|
||||
"Used to work around ulimit problems of CI runner",
|
||||
"Potentially useful for debugging memory problems",
|
||||
"32-bit allocator is always used if your host kernel is older than 4.17"
|
||||
]
|
||||
},
|
||||
"GlobalJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -12,10 +12,6 @@
|
||||
#include <string.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
class SyscallVisitor;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
@@ -26,12 +22,6 @@ namespace FEXCore::Context {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CPUID.Init(this);
|
||||
return true;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
@@ -40,28 +30,12 @@ namespace FEXCore::Context {
|
||||
return CustomExitHandler;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::Stop() {
|
||||
Stop(false);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
|
||||
return ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsDone() const {
|
||||
return IsPaused();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
|
||||
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
|
||||
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
|
||||
@@ -14,8 +14,8 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -37,7 +37,6 @@
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
@@ -45,7 +44,6 @@ namespace CodeSerialize {
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
class Dispatcher;
|
||||
}
|
||||
namespace HLE {
|
||||
@@ -70,35 +68,29 @@ namespace FEXCore::Context {
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct ExitFunctionLinkData {
|
||||
uint64_t HostBranch;
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void(*)(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record);
|
||||
constexpr uint32_t TSC_SCALE = 128;
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitializeContext() override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
|
||||
bool InitCore() override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
ExitHandler GetExitHandler() const override;
|
||||
|
||||
void Pause() override;
|
||||
void Run() override;
|
||||
void Stop() override;
|
||||
void Step() override;
|
||||
|
||||
ExitReason RunUntilExit() override;
|
||||
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
|
||||
|
||||
int GetProgramStatus() const override;
|
||||
|
||||
ExitReason GetExitReason() override;
|
||||
|
||||
bool IsDone() const override;
|
||||
|
||||
void GetCPUState(FEXCore::Core::CPUState *State) const override;
|
||||
void SetCPUState(const FEXCore::Core::CPUState *State) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
@@ -107,55 +99,48 @@ namespace FEXCore::Context {
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
/**
|
||||
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* The OS thread will wait until RunThread is executed
|
||||
*/
|
||||
void InitializeThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
/**
|
||||
* @brief Starts the OS thread object to start executing guest code
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void RunThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void StopThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread, bool NeedsTLSUninstall) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
@@ -187,13 +172,19 @@ namespace FEXCore::Context {
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
FEXCore::ForkableSharedMutex &GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
void MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr);
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
|
||||
@@ -202,9 +193,6 @@ namespace FEXCore::Context {
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
#endif
|
||||
#ifdef JIT_X86_64
|
||||
friend class FEXCore::CPU::X86JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
@@ -215,6 +203,9 @@ namespace FEXCore::Context {
|
||||
// this is for internal use
|
||||
bool ValidateIRarser { false };
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck { false };
|
||||
|
||||
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
|
||||
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
@@ -232,7 +223,6 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
@@ -242,25 +232,16 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
} Config;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
FEXCore::Core::InternalThreadState* ParentThread{};
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
std::mutex IdleWaitMutex;
|
||||
std::condition_variable IdleWaitCV;
|
||||
std::atomic<uint32_t> IdleWaitRefCount{};
|
||||
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
// CPUID depends on HostFeatures so needs to be initialized after that.
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
@@ -280,25 +261,15 @@ namespace FEXCore::Context {
|
||||
ContextImpl();
|
||||
~ContextImpl();
|
||||
|
||||
bool IsPaused() const { return !Running; }
|
||||
void WaitForThreadsToRun();
|
||||
void Stop(bool IgnoreCurrentThread);
|
||||
void WaitForIdle();
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData *HostLink, const BlockDelinkerFunc &delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, ExitFunctionLinkData *Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
return Fn(Frame, Record);
|
||||
}
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
@@ -307,7 +278,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
@@ -322,7 +293,7 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
@@ -333,11 +304,8 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
@@ -349,8 +317,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
@@ -365,10 +331,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void IncrementIdleRefCount() override {
|
||||
++IdleWaitRefCount;
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
@@ -401,8 +363,6 @@ namespace FEXCore::Context {
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
@@ -415,15 +375,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* InitCore and CreateThread both call this to finish up thread object initialization
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
@@ -433,16 +384,8 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
|
||||
// Entry Cache
|
||||
std::mutex ExitMutex;
|
||||
fextl::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
@@ -454,9 +397,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::atomic<bool> HasCustomIRHandlers{};
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
@@ -8,10 +9,11 @@
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
#include <cpu-features.h>
|
||||
@@ -27,8 +29,9 @@ namespace FEXCore::CPU {
|
||||
// TODO: Allow x18 register allocation on Linux in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
@@ -36,23 +39,23 @@ namespace x64 {
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
@@ -80,6 +83,54 @@ namespace x64 {
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
#else
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r0,
|
||||
FEXCore::ARMEmitter::Reg::r1, FEXCore::ARMEmitter::Reg::r27,
|
||||
// SP's register location isn't specified by the ARM64EC ABI, we choose to use r23
|
||||
FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26,
|
||||
FEXCore::ARMEmitter::Reg::r2, FEXCore::ARMEmitter::Reg::r3,
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r20,
|
||||
FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22,
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r14,FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
}};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
#endif
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
@@ -175,19 +226,20 @@ namespace x64 {
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
@@ -199,11 +251,10 @@ namespace x32 {
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
@@ -335,11 +386,11 @@ namespace x32 {
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
|
||||
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
|
||||
, EmitterCTX {ctx}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&SimDecoder}
|
||||
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
|
||||
#endif
|
||||
{
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -352,8 +403,10 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
// Only setup the disassembler if enabled.
|
||||
// vixl's decoder is expensive to setup.
|
||||
if (Disassemble()) {
|
||||
DisasmBuffer.resize(DISASM_BUFFER_SIZE);
|
||||
Disasm = fextl::make_unique<vixl::aarch64::Disassembler>(DisasmBuffer.data(), DISASM_BUFFER_SIZE);
|
||||
DisasmDecoder = fextl::make_unique<vixl::aarch64::Decoder>();
|
||||
DisasmDecoder->AppendVisitor(&Disasm);
|
||||
DisasmDecoder->AppendVisitor(Disasm.get());
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -366,9 +419,12 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
@@ -379,13 +435,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
}
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
@@ -406,6 +455,15 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
|
||||
movn(s, Reg.W(), (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
@@ -581,9 +639,37 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
//
|
||||
// Disable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
|
||||
// is always static and almost certainly clobbered by the subsequent code.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
|
||||
GPRSpillMask &= ~PFAFSpillMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
@@ -600,6 +686,14 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
|
||||
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
@@ -645,21 +739,57 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
[[maybe_unused]] bool FoundRegister{};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & GPRFillMask)) {
|
||||
TmpReg = Reg;
|
||||
FoundRegister = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(FoundRegister, "Didn't have an SRA register to use as a temporary while spilling!");
|
||||
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// Enable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
// is always static and was almost certainly clobbered.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
|
||||
if (FPRs) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE) {
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
@@ -672,8 +802,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
@@ -704,6 +832,11 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
|
||||
GPRFillMask &= ~PFAFMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
@@ -718,6 +851,14 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
|
||||
@@ -842,7 +983,9 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
// Push the general registers.
|
||||
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
@@ -854,7 +997,9 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
// Pop GPRs second
|
||||
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
|
||||
@@ -21,6 +21,7 @@
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -36,16 +37,37 @@ namespace FEXCore::CPU {
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
// GPR temporaries. Only x3 can be used across spill boundaries
|
||||
// so if these ever need to change, be very careful about that.
|
||||
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
|
||||
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
|
||||
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
|
||||
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
|
||||
constexpr bool TMP_ABIARGS = true;
|
||||
|
||||
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
|
||||
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
|
||||
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
|
||||
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
|
||||
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
|
||||
#else
|
||||
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x10;
|
||||
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x11;
|
||||
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x12;
|
||||
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x13;
|
||||
constexpr bool TMP_ABIARGS = false;
|
||||
|
||||
// We pin r11/r12 as PF/AF respectively for arm64ec, as r26/r27 are used for SRA.
|
||||
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r9;
|
||||
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r24;
|
||||
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v16;
|
||||
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
|
||||
#endif
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
@@ -53,12 +75,12 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
|
||||
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
@@ -129,21 +151,21 @@ protected:
|
||||
|
||||
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
SpillForPreserveAllABICall(TMP1, true);
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
}
|
||||
else {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
}
|
||||
}
|
||||
|
||||
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(true);
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
}
|
||||
else {
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -227,15 +249,17 @@ protected:
|
||||
#ifdef VIXL_SIMULATOR
|
||||
vixl::aarch64::Decoder SimDecoder;
|
||||
vixl::aarch64::Simulator Simulator;
|
||||
constexpr static size_t SimulatorStackSize = 8 * 1024 * 1024;
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
vixl::aarch64::Disassembler Disasm;
|
||||
fextl::vector<char> DisasmBuffer;
|
||||
constexpr static int DISASM_BUFFER_SIZE {256};
|
||||
fextl::unique_ptr<vixl::aarch64::Disassembler> Disasm;
|
||||
fextl::unique_ptr<vixl::aarch64::Decoder> DisasmDecoder;
|
||||
|
||||
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
|
||||
#endif
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
}
|
||||
@@ -35,8 +35,10 @@ public:
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
void adr(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADR });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void adr(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADR });
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
@@ -62,8 +64,10 @@ public:
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADRP });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADRP });
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
@@ -105,7 +109,7 @@ public:
|
||||
}
|
||||
}
|
||||
void LongAddressGen(FEXCore::ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN });
|
||||
Label->Insts.emplace_back(SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN });
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
@@ -766,6 +770,21 @@ public:
|
||||
EvaluateIntoFlags(Op, 1, rn);
|
||||
}
|
||||
|
||||
void cfinv() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0001'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
void axflag() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
void xaflag() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
// Conditional compare - register
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
|
||||
@@ -60,7 +60,7 @@ public:
|
||||
}
|
||||
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
|
||||
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Cryptographic two-register SHA
|
||||
|
||||
@@ -18,8 +18,10 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
}
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
@@ -45,8 +47,10 @@ public:
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
}
|
||||
@@ -102,8 +106,10 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void b(ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
@@ -131,8 +137,10 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void bl(ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bl(LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
@@ -163,8 +171,10 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
@@ -195,8 +205,10 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
@@ -226,8 +238,11 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
@@ -256,8 +271,11 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
@@ -4,13 +4,14 @@
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
|
||||
#include <array>
|
||||
@@ -537,27 +538,29 @@ namespace FEXCore::ARMEmitter {
|
||||
uint8_t *Location{};
|
||||
};
|
||||
|
||||
/* This `ForwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
/* This `SingleUseForwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `above` an instruction that uses it.
|
||||
* Which means that a branch would jump forwards.
|
||||
*
|
||||
* This can be bound to multiple instructions, so it needs a vector for each bind instruction type.
|
||||
* The `ForwardLabel` struct can be bound to multiple instructions, so it needs a vector for each bind instruction type.
|
||||
*/
|
||||
struct ForwardLabel {
|
||||
struct Instructions {
|
||||
enum class InstType {
|
||||
ADR,
|
||||
ADRP,
|
||||
B,
|
||||
BC,
|
||||
TEST_BRANCH,
|
||||
RELATIVE_LOAD,
|
||||
LONG_ADDRESS_GEN,
|
||||
};
|
||||
uint8_t *Location{};
|
||||
InstType Type;
|
||||
struct SingleUseForwardLabel {
|
||||
enum class InstType {
|
||||
UNKNOWN,
|
||||
ADR,
|
||||
ADRP,
|
||||
B,
|
||||
BC,
|
||||
TEST_BRANCH,
|
||||
RELATIVE_LOAD,
|
||||
LONG_ADDRESS_GEN,
|
||||
};
|
||||
fextl::vector<Instructions> Insts{};
|
||||
uint8_t *Location{};
|
||||
InstType Type = InstType::UNKNOWN;
|
||||
};
|
||||
|
||||
struct ForwardLabel {
|
||||
fextl::vector<SingleUseForwardLabel> Insts{};
|
||||
};
|
||||
|
||||
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
@@ -569,6 +572,15 @@ namespace FEXCore::ARMEmitter {
|
||||
ForwardLabel Forward;
|
||||
};
|
||||
|
||||
static inline void AddLocationToLabel(SingleUseForwardLabel *Label, SingleUseForwardLabel&& Location) {
|
||||
LOGMAN_THROW_A_FMT(Label->Type == SingleUseForwardLabel::InstType::UNKNOWN, "Trying to bind a SingleUseForwardLabel to multiple locations. Use ForwardLabel instead.");
|
||||
*Label = std::move(Location);
|
||||
}
|
||||
|
||||
static inline void AddLocationToLabel(ForwardLabel *Label, SingleUseForwardLabel&& Location) {
|
||||
Label->Insts.emplace_back(std::move(Location));
|
||||
}
|
||||
|
||||
// Some FCMA ASIMD instructions support a rotation argument.
|
||||
enum class Rotation : uint32_t {
|
||||
ROTATE_0 = 0b00,
|
||||
@@ -628,6 +640,121 @@ namespace FEXCore::ARMEmitter {
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
}
|
||||
|
||||
void Bind(const SingleUseForwardLabel *Label) {
|
||||
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
// Patch up the instructions
|
||||
switch (Label->Type) {
|
||||
case SingleUseForwardLabel::InstType::ADR: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= (Offset & 0b11) << 29;
|
||||
Inst |= (Offset >> 2) << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::ADRP: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= (Offset & 0b11) << 29;
|
||||
Inst |= (Offset >> 2) << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
|
||||
case SingleUseForwardLabel::InstType::B: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FF'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= Offset;
|
||||
*Instruction = Inst;
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case SingleUseForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~(InstMask << 5);
|
||||
Inst |= Offset << 5;
|
||||
*Instruction = Inst;
|
||||
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::BC:
|
||||
case SingleUseForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x7'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~(InstMask << 5);
|
||||
Inst |= Offset << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
auto OriginalOffset = GetCursorOffset();
|
||||
|
||||
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
SetCursorOffset(InstOffset);
|
||||
|
||||
// We encoded the destination register in to the first instruction space.
|
||||
// Read it back.
|
||||
ARMEmitter::Register DestReg(Instructions[0]);
|
||||
|
||||
if (IsADRRange(ImmInstTwo)) {
|
||||
// If within ADR range from the second instruction, then we can emit NOP+ADR
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
|
||||
}
|
||||
else if (IsADRPRange(ImmInstOne)) {
|
||||
|
||||
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
|
||||
// First check if we are in non-offset range for second instruction.
|
||||
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
|
||||
// We can emit nop + adrp
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
|
||||
}
|
||||
else {
|
||||
// Not aligned, need adrp + add
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
|
||||
}
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
SetCursorOffset(OriginalOffset);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
}
|
||||
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
@@ -636,119 +763,8 @@ namespace FEXCore::ARMEmitter {
|
||||
if constexpr (WarnAboutEmpty) {
|
||||
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
|
||||
}
|
||||
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
for (const auto &Inst : Label->Insts) {
|
||||
// Patch up the instructions
|
||||
switch (Inst.Type) {
|
||||
case ForwardLabel::Instructions::InstType::ADR: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= (Offset & 0b11) << 29;
|
||||
Inst |= (Offset >> 2) << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::Instructions::InstType::ADRP: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= (Offset & 0b11) << 29;
|
||||
Inst |= (Offset >> 2) << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::Instructions::InstType::B: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FF'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
Inst |= Offset;
|
||||
*Instruction = Inst;
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::Instructions::InstType::TEST_BRANCH: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~(InstMask << 5);
|
||||
Inst |= Offset << 5;
|
||||
*Instruction = Inst;
|
||||
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::Instructions::InstType::BC:
|
||||
case ForwardLabel::Instructions::InstType::RELATIVE_LOAD: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x7'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
uint32_t Inst = *Instruction & ~(InstMask << 5);
|
||||
Inst |= Offset << 5;
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
auto OriginalOffset = GetCursorOffset();
|
||||
|
||||
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
SetCursorOffset(InstOffset);
|
||||
|
||||
// We encoded the destination register in to the first instruction space.
|
||||
// Read it back.
|
||||
ARMEmitter::Register DestReg(Instructions[0]);
|
||||
|
||||
if (IsADRRange(ImmInstTwo)) {
|
||||
// If within ADR range from the second instruction, then we can emit NOP+ADR
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
|
||||
}
|
||||
else if (IsADRPRange(ImmInstOne)) {
|
||||
|
||||
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
|
||||
// First check if we are in non-offset range for second instruction.
|
||||
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
|
||||
// We can emit nop + adrp
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
|
||||
}
|
||||
else {
|
||||
// Not aligned, need adrp + add
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
|
||||
}
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
SetCursorOffset(OriginalOffset);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
for (auto &Inst : Label->Insts) {
|
||||
Bind(&Inst);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -2121,38 +2121,58 @@ public:
|
||||
LoadStoreLiteral(Op, prfop, static_cast<uint32_t>(Imm >> 2) & 0x7'FFFF);
|
||||
}
|
||||
|
||||
void ldr(FEXCore::ARMEmitter::WRegister rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void ldr(FEXCore::ARMEmitter::WRegister rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b0001'1000 << 24;
|
||||
LoadStoreLiteral(Op, rt, 0);
|
||||
}
|
||||
void ldr(FEXCore::ARMEmitter::SRegister rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void ldr(FEXCore::ARMEmitter::SRegister rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b0001'1100 << 24;
|
||||
LoadStoreLiteral(Op, rt, 0);
|
||||
}
|
||||
void ldr(FEXCore::ARMEmitter::XRegister rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void ldr(FEXCore::ARMEmitter::XRegister rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b0101'1000 << 24;
|
||||
LoadStoreLiteral(Op, rt, 0);
|
||||
}
|
||||
void ldr(FEXCore::ARMEmitter::DRegister rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void ldr(FEXCore::ARMEmitter::DRegister rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b0101'1100 << 24;
|
||||
LoadStoreLiteral(Op, rt, 0);
|
||||
}
|
||||
void ldrsw(FEXCore::ARMEmitter::XRegister rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void ldrsw(FEXCore::ARMEmitter::XRegister rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b1001'1000 << 24;
|
||||
LoadStoreLiteral(Op, rt, 0);
|
||||
}
|
||||
void ldr(FEXCore::ARMEmitter::QRegister rt, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void ldr(FEXCore::ARMEmitter::QRegister rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b1001'1100 << 24;
|
||||
LoadStoreLiteral(Op, rt, 0);
|
||||
}
|
||||
void prfm(FEXCore::ARMEmitter::Prefetch prfop, ForwardLabel *Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void prfm(FEXCore::ARMEmitter::Prefetch prfop, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
|
||||
constexpr uint32_t Op = 0b1101'1000 << 24;
|
||||
LoadStoreLiteral(Op, prfop, 0);
|
||||
}
|
||||
|
||||
@@ -1384,12 +1384,10 @@ public:
|
||||
}
|
||||
|
||||
// SVE predicate initialize
|
||||
template <SubRegSize size>
|
||||
void ptrue(PRegister pd, PredicatePattern pattern) {
|
||||
void ptrue(SubRegSize size, PRegister pd, PredicatePattern pattern) {
|
||||
SVEPredicateMisc(0b1000, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
|
||||
}
|
||||
template <SubRegSize size>
|
||||
void ptrues(PRegister pd, PredicatePattern pattern) {
|
||||
void ptrues(SubRegSize size, PRegister pd, PredicatePattern pattern) {
|
||||
SVEPredicateMisc(0b1001, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
|
||||
}
|
||||
|
||||
|
||||
@@ -798,6 +798,52 @@ public:
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
void fmov(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000000, rd, rn);
|
||||
}
|
||||
void fabs(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000001, rd, rn);
|
||||
}
|
||||
void fneg(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000010, rd, rn);
|
||||
}
|
||||
void fsqrt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000011, rd, rn);
|
||||
}
|
||||
void frintn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001000, rd, rn);
|
||||
}
|
||||
void frintp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001001, rd, rn);
|
||||
}
|
||||
void frintm(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001010, rd, rn);
|
||||
}
|
||||
void frintz(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001011, rd, rn);
|
||||
}
|
||||
void frinta(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001100, rd, rn);
|
||||
}
|
||||
void frintx(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001110, rd, rn);
|
||||
}
|
||||
void frinti(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001111, rd, rn);
|
||||
}
|
||||
void frint32z(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010000, rd, rn);
|
||||
}
|
||||
void frint32x(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010001, rd, rn);
|
||||
}
|
||||
void frint64z(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010010, rd, rn);
|
||||
}
|
||||
void frint64x(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010011, rd, rn);
|
||||
}
|
||||
|
||||
void fmov(SRegister rd, SRegister rn) {
|
||||
Float1Source(0, 0, 0b00, 0b000000, rd.V(), rn.V());
|
||||
}
|
||||
@@ -1065,6 +1111,34 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point data-processing (2 source)
|
||||
void fmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0000, rd, rn, rm);
|
||||
}
|
||||
void fdiv(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0001, rd, rn, rm);
|
||||
}
|
||||
void fadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0010, rd, rn, rm);
|
||||
}
|
||||
void fsub(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0011, rd, rn, rm);
|
||||
}
|
||||
void fmax(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0100, rd, rn, rm);
|
||||
}
|
||||
void fmin(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0101, rd, rn, rm);
|
||||
}
|
||||
void fmaxnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0110, rd, rn, rm);
|
||||
}
|
||||
void fminnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0111, rd, rn, rm);
|
||||
}
|
||||
void fnmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b1000, rd, rn, rm);
|
||||
}
|
||||
|
||||
void fmul(SRegister rd, SRegister rn, SRegister rm) {
|
||||
Float2Source(0, 0, 0b00, 0b0000, rd.V(), rn.V(), rm.V());
|
||||
}
|
||||
@@ -1150,6 +1224,16 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
void fcsel(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
FloatConditionalSelect(0, 0, ConvertedSize, rd, rn, rm, Cond);
|
||||
}
|
||||
|
||||
void fcsel(SRegister rd, SRegister rn, SRegister rm, Condition Cond) {
|
||||
FloatConditionalSelect(0, 0, 0b00, rd.V(), rn.V(), rm.V(), Cond);
|
||||
}
|
||||
@@ -1305,6 +1389,16 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
void Float1Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float1Source(M, S, ConvertedSize, opcode, rd, rn);
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
void FloatCompare(uint32_t M, uint32_t S, uint32_t ftype, uint32_t op, uint32_t opcode2, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0010'0000'0000'0000;
|
||||
@@ -1337,6 +1431,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point data-processing (2 source)
|
||||
|
||||
void Float2Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1351,6 +1446,16 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void Float2Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float2Source(M, S, ConvertedSize, opcode, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
void FloatConditionalSelect(uint32_t M, uint32_t S, uint32_t ptype, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1100'0000'0000;
|
||||
|
||||
@@ -5,6 +5,10 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
@@ -17,6 +21,12 @@ constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant:
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
|
||||
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
|
||||
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
|
||||
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
|
||||
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {
|
||||
@@ -176,6 +186,96 @@ constexpr static auto SHUFPS_LUT {
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPS_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint32_t Val[4];
|
||||
};
|
||||
|
||||
std::array<LUTType, 16> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1U;
|
||||
}
|
||||
return 0U;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
LUT.Val[2] = GetLUT(i, 2);
|
||||
LUT.Val[3] = GetLUT(i, 3);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPD_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
|
||||
std::array<LUTType, 4> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1ULL;
|
||||
}
|
||||
return 0ULL;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto PBLENDW_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint16_t Val[8];
|
||||
};
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
|
||||
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
|
||||
// PBLENDW behaviour:
|
||||
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
|
||||
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
|
||||
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
|
||||
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
|
||||
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
|
||||
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
|
||||
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
|
||||
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
|
||||
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
const uint16_t WordSelectionSrc[8] = {
|
||||
0x01'00,
|
||||
0x03'02,
|
||||
0x05'04,
|
||||
0x07'06,
|
||||
0x09'08,
|
||||
0x0B'0A,
|
||||
0x0D'0C,
|
||||
0x0F'0E,
|
||||
};
|
||||
|
||||
constexpr uint16_t OriginalDest = 0xFF'FF;
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
for (size_t j = 0; j < 8; ++j) {
|
||||
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
|
||||
}
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
|
||||
|
||||
@@ -194,6 +294,9 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] = reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
// Fill in telemetry values
|
||||
@@ -250,6 +353,31 @@ auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
//
|
||||
// MDWE prevents applications from creating RWX memory mappings.
|
||||
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
|
||||
//
|
||||
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
|
||||
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
|
||||
//
|
||||
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
|
||||
//
|
||||
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
|
||||
// -1: The kernel doesn't support MDWE
|
||||
// 0: MDWE is supported but disabled
|
||||
// >0: MDWE is enabled, hence prohibiting RWX mappings
|
||||
#ifndef PR_GET_MDWE
|
||||
#define PR_GET_MDWE 66
|
||||
#endif
|
||||
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
|
||||
if (MDWE != -1 && MDWE != 0) {
|
||||
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
|
||||
}
|
||||
#endif
|
||||
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
|
||||
@@ -33,14 +33,19 @@ namespace ProductNames {
|
||||
static const char ARM_A76[] = "Cortex-A76";
|
||||
static const char ARM_A76AE[] = "Cortex-A76AE";
|
||||
static const char ARM_V1[] = "Neoverse V1";
|
||||
static const char ARM_V2[] = "Neoverse V2";
|
||||
static const char ARM_A77[] = "Cortex-A77";
|
||||
static const char ARM_A78[] = "Cortex-A78";
|
||||
static const char ARM_A78AE[] = "Cortex-A78AE";
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_E1[] = "Neoverse E1";
|
||||
@@ -49,6 +54,7 @@ namespace ProductNames {
|
||||
static const char ARM_A55[] = "Cortex-A55";
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
static const char ARM_A520[] = "Cortex-A520";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
@@ -92,7 +98,7 @@ constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
uint32_t GetCycleCounterFrequency() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], CNTFRQ_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
@@ -100,11 +106,10 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
PerCPUData.resize(Cores);
|
||||
|
||||
uint64_t MIDR{};
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec{};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
@@ -138,10 +143,15 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 36> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 42> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
@@ -173,6 +183,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
{0x41, 0xd05, 0, ProductNames::ARM_A55}, // A55
|
||||
@@ -206,7 +217,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
fextl::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
if (MIDROption) {
|
||||
@@ -322,7 +333,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
}
|
||||
else {
|
||||
// If we aren't hybrid then just claim everything is big
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
|
||||
@@ -338,7 +349,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
uint32_t GetCycleCounterFrequency() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -347,6 +358,34 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
void CPUIDEmu::SetupFeatures() {
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(CPUIDFeatures, CPUID);
|
||||
if (!CPUIDFeatures()) {
|
||||
// Early exit if no features are overriden.
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features.FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features.FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
ENABLE_DISABLE_OPTION(SHA, SHA, SHA);
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
@@ -368,7 +407,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
// Processor Info and Features bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -377,7 +415,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(CoreCount << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
|
||||
Res.ecx =
|
||||
@@ -484,7 +522,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
|
||||
if (Leaf == 0) {
|
||||
// Report L1D
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Data | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
@@ -508,7 +546,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 1) {
|
||||
// Report L1I
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Instruction | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
@@ -532,7 +570,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
// Report L2
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b010 << 5) | // Cache level
|
||||
@@ -556,7 +594,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
@@ -629,7 +667,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // SHA instructions
|
||||
(Features.SHA << 29) | // SHA instructions
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
@@ -765,7 +803,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
if (FrequencyHz) {
|
||||
Res.eax = 1;
|
||||
Res.ebx = 1;
|
||||
Res.ebx = CTX->Config.SmallTSCScale() ? FEXCore::Context::TSC_SCALE : 1;
|
||||
Res.ecx = FrequencyHz;
|
||||
}
|
||||
return Res;
|
||||
@@ -1058,7 +1096,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
|
||||
(0 << 1) | // IRPerf: Instructions retired count support
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
Res.ecx =
|
||||
(0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
@@ -1156,7 +1194,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) con
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
@@ -1195,16 +1233,14 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
return Res;
|
||||
}
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
CTX = ctx;
|
||||
CPUIDEmu::CPUIDEmu(FEXCore::Context::ContextImpl const *ctx)
|
||||
: CTX {ctx} {
|
||||
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
SetupFeatures();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,6 +14,8 @@ namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
uint32_t GetCycleCounterFrequency();
|
||||
|
||||
// Debugging define to switch what family of CPU we execute as.
|
||||
// Might be useful if an application makes an assumption about a CPU.
|
||||
// #define CPUID_AMD
|
||||
@@ -28,12 +30,12 @@ private:
|
||||
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
|
||||
|
||||
public:
|
||||
CPUIDEmu(FEXCore::Context::ContextImpl const *ctx);
|
||||
|
||||
// X86 cacheline size effectively has to be hardcoded to 64
|
||||
// if we report anything differently then applications are likely to break
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
|
||||
void Init(FEXCore::Context::ContextImpl *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) const {
|
||||
if (Function < Primary.size()) {
|
||||
const auto Handler = Primary[Function];
|
||||
@@ -111,10 +113,11 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::ContextImpl const *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
uint32_t Cores{};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
@@ -136,6 +139,15 @@ private:
|
||||
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
|
||||
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
|
||||
|
||||
struct FeaturesConfig {
|
||||
uint64_t SHA : 1;
|
||||
uint64_t _pad : 63;
|
||||
};
|
||||
|
||||
FeaturesConfig Features {
|
||||
.SHA = 1,
|
||||
};
|
||||
|
||||
uint64_t XCR0 {
|
||||
XCR0_X87 |
|
||||
XCR0_SSE
|
||||
@@ -189,6 +201,7 @@ private:
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h() const;
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
void SetupFeatures();
|
||||
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
|
||||
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
|
||||
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
|
||||
|
||||
@@ -9,18 +9,19 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
@@ -39,13 +40,13 @@ $end_info$
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "FEXCore/Utils/SignalScopeGuards.h"
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -76,67 +77,10 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
};
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
"PF",
|
||||
"",
|
||||
"AF",
|
||||
"",
|
||||
"ZF",
|
||||
"SF",
|
||||
"TF",
|
||||
"IF",
|
||||
"DF",
|
||||
"OF",
|
||||
"IOPL",
|
||||
"",
|
||||
"NT",
|
||||
"",
|
||||
"RF",
|
||||
"VM",
|
||||
"AC",
|
||||
"VIF",
|
||||
"VIP",
|
||||
"ID",
|
||||
};
|
||||
|
||||
std::string_view const& GetFlagName(unsigned Flag) {
|
||||
return FlagNames[Flag];
|
||||
}
|
||||
|
||||
constexpr std::array<std::string_view const, 16> RegNames = {
|
||||
"rax",
|
||||
"rbx",
|
||||
"rcx",
|
||||
"rdx",
|
||||
"rsi",
|
||||
"rdi",
|
||||
"rbp",
|
||||
"rsp",
|
||||
"r8",
|
||||
"r9",
|
||||
"r10",
|
||||
"r11",
|
||||
"r12",
|
||||
"r13",
|
||||
"r14",
|
||||
"r15",
|
||||
};
|
||||
|
||||
std::string_view const& GetGRegName(unsigned Reg) {
|
||||
return RegNames[Reg];
|
||||
}
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl()
|
||||
: IRCaptureCache {this} {
|
||||
: CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
@@ -155,30 +99,19 @@ namespace FEXCore::Context {
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
if (FEXCore::GetCycleCounterFrequency() >= FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
Config.SmallTSCScale = false;
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
if (ParentThread) {
|
||||
DestroyThread(ParentThread);
|
||||
}
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
CodeObjectCacheService->Shutdown();
|
||||
}
|
||||
|
||||
for (auto &Thread : Threads) {
|
||||
if (Thread->ExecutionThread->joinable()) {
|
||||
Thread->ExecutionThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
for (auto &Thread : Threads) {
|
||||
delete Thread;
|
||||
}
|
||||
Threads.clear();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -221,45 +154,66 @@ namespace FEXCore::Context {
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) {
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
uint32_t EFLAGS{};
|
||||
|
||||
// Currently these flags just map 1:1 inside of the resulting value.
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
if (i == X86State::RFLAG_PF_LOC || i == X86State::RFLAG_AF_LOC) {
|
||||
// Intentionally do nothing.
|
||||
// These contain multiple bits which can corrupt other members when compacted.
|
||||
continue;
|
||||
switch (i) {
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
// Intentionally do nothing.
|
||||
// These contain multiple bits which can corrupt other members when compacted.
|
||||
break;
|
||||
default:
|
||||
EFLAGS |= uint32_t{Frame->State.flags[i]} << i;
|
||||
break;
|
||||
}
|
||||
|
||||
EFLAGS |= uint32_t{Frame->State.flags[i]} << i;
|
||||
}
|
||||
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
uint32_t Packed_NZCV{};
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_LOC)) & 1;
|
||||
if (WasInJIT) {
|
||||
// If we were in the JIT then NZCV is in the CPU's PSTATE object.
|
||||
// Packed in to the same bit locations as RFLAG_NZCV_LOC.
|
||||
Packed_NZCV = PSTATE;
|
||||
|
||||
// If we were in the JIT then PF and AF are in registers.
|
||||
// Move them to the CPUState frame now.
|
||||
Frame->State.pf_raw = HostGPRs[CPU::REG_PF.Idx()];
|
||||
Frame->State.af_raw = HostGPRs[CPU::REG_AF.Idx()];
|
||||
}
|
||||
else {
|
||||
// If we were not in the JIT then the NZCV state is stored in the CPUState RFLAG_NZCV_LOC.
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
}
|
||||
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC)) & 1;
|
||||
|
||||
// Pack in to EFLAGS
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_LOC;
|
||||
EFLAGS |= ZF << X86State::RFLAG_ZF_LOC;
|
||||
EFLAGS |= SF << X86State::RFLAG_SF_LOC;
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_RAW_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_RAW_LOC;
|
||||
EFLAGS |= ZF << X86State::RFLAG_ZF_RAW_LOC;
|
||||
EFLAGS |= SF << X86State::RFLAG_SF_RAW_LOC;
|
||||
|
||||
// PF calculation is deferred, calculate it now.
|
||||
// Popcount the 8-bit flag and then extract the lower bit.
|
||||
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_LOC];
|
||||
uint32_t PFByte = Frame->State.pf_raw & 0xff;
|
||||
uint32_t PF = std::popcount(PFByte ^ 1) & 1;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_LOC;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_RAW_LOC;
|
||||
|
||||
// AF calculation is deferred, calculate it now.
|
||||
// XOR with PF byte and extract bit 4.
|
||||
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_LOC;
|
||||
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
return EFLAGS;
|
||||
}
|
||||
@@ -268,21 +222,21 @@ namespace FEXCore::Context {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
switch (i) {
|
||||
case X86State::RFLAG_OF_LOC:
|
||||
case X86State::RFLAG_CF_LOC:
|
||||
case X86State::RFLAG_ZF_LOC:
|
||||
case X86State::RFLAG_SF_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
// Intentionally do nothing.
|
||||
break;
|
||||
case X86State::RFLAG_AF_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
// AF stored in bit 4 in our internal representation. It is also
|
||||
// XORed with byte 4 of the PF byte, but we write that as zero here so
|
||||
// we don't need any special handling for that.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
Frame->State.af_raw = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
break;
|
||||
case X86State::RFLAG_PF_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
// PF is inverted in our internal representation.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
Frame->State.pf_raw = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
break;
|
||||
default:
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 1 : 0;
|
||||
@@ -292,10 +246,10 @@ namespace FEXCore::Context {
|
||||
|
||||
// Calculate packed NZCV
|
||||
uint32_t Packed_NZCV{};
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC) : 0;
|
||||
memcpy(&Frame->State.flags[X86State::RFLAG_NZCV_LOC], &Packed_NZCV, sizeof(Packed_NZCV));
|
||||
|
||||
// Reserved, Read-As-1, Write-as-1
|
||||
@@ -304,7 +258,7 @@ namespace FEXCore::Context {
|
||||
Frame->State.flags[X86State::RFLAG_IF_LOC] = 1;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
bool ContextImpl::InitCore() {
|
||||
// Initialize the CPU core signal handlers & DispatcherConfig
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
@@ -314,21 +268,20 @@ namespace FEXCore::Context {
|
||||
// Do nothing
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown core configuration");
|
||||
break;
|
||||
LogMan::Msg::EFmt("Unknown core configuration");
|
||||
return false;
|
||||
}
|
||||
|
||||
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this, DispatcherConfig);
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.StaticRegisterAllocation = DispatcherConfig.StaticRegisterAllocation,
|
||||
.SupportsAVX = HostFeatures.SupportsAVX,
|
||||
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
|
||||
.AbsoluteLoopTopAddress = Dispatcher->AbsoluteLoopTopAddress,
|
||||
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
|
||||
@@ -352,207 +305,31 @@ namespace FEXCore::Context {
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
else {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
// WIN32 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
InitializeThreadData(Thread);
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::StartGdbServer() {
|
||||
#ifndef _WIN32
|
||||
if (!DebugServer) {
|
||||
DebugServer = fextl::make_unique<GdbServer>(this);
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
// FEX needs to start paused when gdb is enabled.
|
||||
StartPaused = true;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::StopGdbServer() {
|
||||
#ifndef _WIN32
|
||||
DebugServer.reset();
|
||||
#endif
|
||||
return true;
|
||||
}
|
||||
|
||||
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForIdle() {
|
||||
std::unique_lock<std::mutex> lk(IdleWaitMutex);
|
||||
IdleWaitCV.wait(lk, [this] {
|
||||
return IdleWaitRefCount.load() == 0;
|
||||
});
|
||||
|
||||
Running = false;
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForIdleWithTimeout() {
|
||||
std::unique_lock<std::mutex> lk(IdleWaitMutex);
|
||||
bool WaitResult = IdleWaitCV.wait_for(lk, std::chrono::milliseconds(1500),
|
||||
[this] {
|
||||
return IdleWaitRefCount.load() == 0;
|
||||
});
|
||||
|
||||
if (!WaitResult) {
|
||||
// The wait failed, this will occur if we stepped in to a syscall
|
||||
// That's okay, we just need to pause the threads manually
|
||||
NotifyPause();
|
||||
}
|
||||
|
||||
// We have sent every thread a pause signal
|
||||
// Now wait again because they /will/ be going to sleep
|
||||
WaitForIdle();
|
||||
}
|
||||
|
||||
void ContextImpl::NotifyPause() {
|
||||
|
||||
// Tell all the threads that they should pause
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Pause);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::Pause() {
|
||||
// If we aren't running, WaitForIdle will never compete.
|
||||
if (Running) {
|
||||
NotifyPause();
|
||||
|
||||
WaitForIdle();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::Run() {
|
||||
// Spin up all the threads
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Return);
|
||||
}
|
||||
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForThreadsToRun() {
|
||||
size_t NumThreads{};
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
NumThreads = Threads.size();
|
||||
}
|
||||
|
||||
// Spin while waiting for the threads to start up
|
||||
std::unique_lock<std::mutex> lk(IdleWaitMutex);
|
||||
IdleWaitCV.wait(lk, [this, NumThreads] {
|
||||
return IdleWaitRefCount.load() >= NumThreads;
|
||||
});
|
||||
|
||||
Running = true;
|
||||
}
|
||||
|
||||
void ContextImpl::Step() {
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
// Walk the threads and tell them to clear their caches
|
||||
// Useful when our block size is set to a large number and we need to step a single instruction
|
||||
for (auto &Thread : Threads) {
|
||||
ClearCodeCache(Thread);
|
||||
}
|
||||
}
|
||||
CoreRunningMode PreviousRunningMode = this->Config.RunningMode;
|
||||
int64_t PreviousMaxIntPerBlock = this->Config.MaxInstPerBlock;
|
||||
this->Config.RunningMode = FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP;
|
||||
this->Config.MaxInstPerBlock = 1;
|
||||
Run();
|
||||
WaitForThreadsToRun();
|
||||
WaitForIdle();
|
||||
this->Config.RunningMode = PreviousRunningMode;
|
||||
this->Config.MaxInstPerBlock = PreviousMaxIntPerBlock;
|
||||
}
|
||||
|
||||
void ContextImpl::Stop(bool IgnoreCurrentThread) {
|
||||
pid_t tid = FHU::Syscalls::gettid();
|
||||
FEXCore::Core::InternalThreadState* CurrentThread{};
|
||||
|
||||
// Tell all the threads that they should stop
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
if (IgnoreCurrentThread &&
|
||||
Thread->ThreadManager.TID == tid) {
|
||||
// If we are callign stop from the current thread then we can ignore sending signals to this thread
|
||||
// This means that this thread is already gone
|
||||
continue;
|
||||
}
|
||||
else if (Thread->ThreadManager.TID == tid) {
|
||||
// We need to save the current thread for last to ensure all threads receive their stop signals
|
||||
CurrentThread = Thread;
|
||||
continue;
|
||||
}
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
StopThread(Thread);
|
||||
}
|
||||
|
||||
// If the thread is waiting to start but immediately killed then there can be a hang
|
||||
// This occurs in the case of gdb attach with immediate kill
|
||||
if (Thread->RunningEvents.WaitingToStart.load()) {
|
||||
Thread->RunningEvents.EarlyExit = true;
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Stop the current thread now if we aren't ignoring it
|
||||
if (CurrentThread) {
|
||||
StopThread(CurrentThread);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->RunningEvents.Running.exchange(false)) {
|
||||
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Stop);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
SignalDelegation->SignalThread(Thread, Event);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason ContextImpl::RunUntilExit() {
|
||||
if(!StartPaused) {
|
||||
// We will only have one thread at this point, but just in case run notify everything
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
|
||||
ExecutionThread(ParentThread);
|
||||
FEXCore::Context::ExitReason ContextImpl::RunUntilExit(FEXCore::Core::InternalThreadState *Thread) {
|
||||
ExecutionThread(Thread);
|
||||
while(true) {
|
||||
this->WaitForIdle();
|
||||
auto reason = ParentThread->ExitReason;
|
||||
auto reason = Thread->ExitReason;
|
||||
|
||||
// Don't return if a custom exit handling the exit
|
||||
if (!CustomExitHandler || reason == ExitReason::EXIT_SHUTDOWN) {
|
||||
@@ -565,69 +342,18 @@ namespace FEXCore::Context {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
}
|
||||
|
||||
int ContextImpl::GetProgramStatus() const {
|
||||
return ParentThread->StatusCode;
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CPUBackend->Initialize();
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
ContextImpl *This;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
static void *ThreadHandler(void* Data) {
|
||||
ExecutionThreadHandler *Handler = reinterpret_cast<ExecutionThreadHandler*>(Data);
|
||||
Handler->This->ExecutionThread(Handler->Thread);
|
||||
FEXCore::Allocator::free(Handler);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// This will create the execution thread but it won't actually start executing
|
||||
ExecutionThreadHandler *Arg = reinterpret_cast<ExecutionThreadHandler*>(FEXCore::Allocator::malloc(sizeof(ExecutionThreadHandler)));
|
||||
Arg->This = this;
|
||||
Arg->Thread = Thread;
|
||||
Thread->StartPaused = NeedToCheckXID;
|
||||
Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
|
||||
// Wait for the thread to have started
|
||||
Thread->ThreadWaiting.Wait();
|
||||
|
||||
if (NeedToCheckXID) {
|
||||
// The first time an application creates a thread, GLIBC installs their SETXID signal handler.
|
||||
// FEX needs to capture all signals and defer them to the guest.
|
||||
// Once FEX creates its first guest thread, overwrite the GLIBC SETXID handler *again* to ensure
|
||||
// FEX maintains control of the signal handler on this signal.
|
||||
NeedToCheckXID = false;
|
||||
SignalDelegation->CheckXIDHandler();
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
|
||||
if (Config.BlockJITNaming() ||
|
||||
Config.GlobalJITNaming() ||
|
||||
Config.LibraryJITNaming()) {
|
||||
// Allocate a TLS JIT symbol buffer only if enabled.
|
||||
Thread->SymbolBuffer = JITSymbols::AllocateBuffer();
|
||||
}
|
||||
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Tell the thread to start executing
|
||||
Thread->StartRunning.NotifyAll();
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
@@ -636,9 +362,6 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->PassManager->RegisterExitHandler([this]() {
|
||||
Stop(false /* Ignore current thread */);
|
||||
});
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
@@ -647,9 +370,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->CTX = this;
|
||||
|
||||
bool DoSRA = DispatcherConfig.StaticRegisterAllocation;
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
|
||||
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
@@ -657,7 +378,7 @@ namespace FEXCore::Context {
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
Thread->PassManager->InsertRegisterAllocationPass(DoSRA, HostFeatures.SupportsAVX);
|
||||
Thread->PassManager->InsertRegisterAllocationPass(HostFeatures.SupportsAVX);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM:
|
||||
@@ -671,48 +392,41 @@ namespace FEXCore::Context {
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
Thread->ThreadManager.parent_tid = ParentTID;
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
|
||||
// Insert after the Thread object has been fully initialized
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
Threads.push_back(Thread);
|
||||
if (Config.BlockJITNaming() ||
|
||||
Config.GlobalJITNaming() ||
|
||||
Config.LibraryJITNaming()) {
|
||||
// Allocate a JIT symbol buffer only if enabled.
|
||||
Thread->SymbolBuffer = JITSymbols::AllocateBuffer();
|
||||
}
|
||||
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// remove new thread object
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
|
||||
auto It = std::find(Threads.begin(), Threads.end(), Thread);
|
||||
LOGMAN_THROW_A_FMT(It != Threads.end(), "Thread wasn't in Threads");
|
||||
|
||||
Threads.erase(It);
|
||||
}
|
||||
|
||||
if (Thread->ExecutionThread &&
|
||||
Thread->ExecutionThread->IsSelf()) {
|
||||
// To be able to delete a thread from itself, we need to detached the std::thread object
|
||||
Thread->ExecutionThread->detach();
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState *Thread, bool NeedsTLSUninstall) {
|
||||
if (NeedsTLSUninstall) {
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
|
||||
@@ -730,43 +444,6 @@ namespace FEXCore::Context {
|
||||
CodeInvalidationMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto &DeadThread : Threads) {
|
||||
if (DeadThread == LiveThread) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Setting running to false ensures that when they are shutdown we won't send signals to kill them
|
||||
DeadThread->RunningEvents.Running = false;
|
||||
|
||||
// Despite what google searches may susgest, glibc actually has special code to handle forks
|
||||
// with multiple active threads.
|
||||
// It cleans up the stacks of dead threads and marks them as terminated.
|
||||
// It also cleans up a bunch of internal mutexes.
|
||||
|
||||
// FIXME: TLS is probally still alive. Investigate
|
||||
|
||||
// Deconstructing the Interneal thread state should clean up most of the state.
|
||||
// But if anything on the now deleted stack is holding a refrence to the heap, it will be leaked
|
||||
delete DeadThread;
|
||||
|
||||
// FIXME: Make sure sure nothing gets leaked via the heap. Ideas:
|
||||
// * Make sure nothing is allocated on the heap without ref in InternalThreadState
|
||||
// * Surround any code that heap allocates with a per-thread mutex.
|
||||
// Before forking, the the forking thread can lock all thread mutexes.
|
||||
}
|
||||
|
||||
// Remove all threads but the live thread from Threads
|
||||
Threads.clear();
|
||||
Threads.push_back(LiveThread);
|
||||
|
||||
// We now only have one thread
|
||||
IdleWaitRefCount = 1;
|
||||
|
||||
// Clean up dead stacks
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -826,7 +503,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
@@ -835,23 +512,26 @@ namespace FEXCore::Context {
|
||||
uint64_t TotalInstructions {0};
|
||||
uint64_t TotalInstructionsLength {0};
|
||||
|
||||
bool HasCustomIR{};
|
||||
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
if (HasCustomIRHandlers.load(std::memory_order_relaxed)) {
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
if (Handler != CustomIRHandlers.end()) {
|
||||
TotalInstructions = 1;
|
||||
TotalInstructionsLength = 1;
|
||||
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
|
||||
HasCustomIR = true;
|
||||
}
|
||||
}
|
||||
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
if (Handler != CustomIRHandlers.end()) {
|
||||
TotalInstructions = 1;
|
||||
TotalInstructionsLength = 1;
|
||||
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
|
||||
lk.unlock();
|
||||
} else {
|
||||
lk.unlock();
|
||||
if (!HasCustomIR) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
|
||||
bool HadDispatchError {false};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
}
|
||||
@@ -1005,7 +685,7 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
@@ -1056,7 +736,7 @@ namespace FEXCore::Context {
|
||||
|
||||
if (IRList == nullptr) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols());
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -1077,7 +757,7 @@ namespace FEXCore::Context {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get(), GetGdbServerStatus()).BlockEntry,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get()).BlockEntry,
|
||||
.IRData = IRList,
|
||||
.DebugData = DebugData,
|
||||
.RAData = std::move(RAData),
|
||||
@@ -1087,23 +767,12 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
void ContextImpl::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto NewBlock = CompileBlock(Frame, GuestRIP);
|
||||
|
||||
if (NewBlock == 0) {
|
||||
LogMan::Msg::EFmt("CompileBlockJit: Failed to compile code {:X} - aborting process", GuestRIP);
|
||||
// Return similar behaviour of SIGILL abort
|
||||
Frame->Thread->StatusCode = 128 + SIGILL;
|
||||
Stop(false /* Ignore current thread */);
|
||||
}
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1118,7 +787,7 @@ namespace FEXCore::Context {
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
@@ -1202,16 +871,11 @@ namespace FEXCore::Context {
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
// Now notify the thread that we are initialized
|
||||
Thread->ThreadWaiting.NotifyAll();
|
||||
|
||||
if (Thread != static_cast<ContextImpl*>(Thread->CTX)->ParentThread || StartPaused || Thread->StartPaused) {
|
||||
if (StartPaused || Thread->StartPaused) {
|
||||
// Parent thread doesn't need to wait to run
|
||||
Thread->StartRunning.Wait();
|
||||
}
|
||||
@@ -1246,18 +910,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
if (!Thread->DestroyedByParent && Thread != static_cast<ContextImpl*>(Thread->CTX)->ParentThread) {
|
||||
Thread->CTX->DestroyThread(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
@@ -1274,44 +929,21 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
static void InvalidateGuestCodeRangeInternal(ContextImpl *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(static_cast<ContextImpl*>(CTX)->ThreadCreationMutex);
|
||||
|
||||
for (auto &Thread : static_cast<ContextImpl*>(CTX)->Threads) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared() {
|
||||
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
LogMan::Throw::AFmt(Threads.size() == 1, "First MarkMemoryShared called must be before creating any threads");
|
||||
|
||||
auto Thread = Threads[0];
|
||||
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
|
||||
std::lock_guard<std::recursive_mutex> lkLookupCache(Thread->LookupCache->WriteLock);
|
||||
Thread->LookupCache->ClearCache();
|
||||
@@ -1322,8 +954,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData *HostLink, const FEXCore::Context::BlockDelinkerFunc &delinker) {
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
@@ -1334,7 +966,7 @@ namespace FEXCore::Context {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->DebugStore.erase(GuestRIP);
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator, void *Data) {
|
||||
@@ -1343,6 +975,7 @@ namespace FEXCore::Context {
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
|
||||
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, std::tuple(Handler, Creator, Data));
|
||||
HasCustomIRHandlers = true;
|
||||
|
||||
if (!InsertedIterator.second) {
|
||||
const auto &[fn, Creator, Data] = InsertedIterator.first->second;
|
||||
@@ -1361,27 +994,17 @@ namespace FEXCore::Context {
|
||||
InvalidateGuestCodeRange(nullptr, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
});
|
||||
}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) {
|
||||
uint64_t Result{};
|
||||
Result = Handler->HandleSyscall(Frame, Args);
|
||||
return Result;
|
||||
HasCustomIRHandlers = !CustomIRHandlers.empty();
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
|
||||
@@ -5,12 +5,14 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
@@ -23,45 +25,25 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Dispatcher::SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
--ctx->IdleWaitRefCount;
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
|
||||
Thread->RunningEvents.ThreadSleeping = true;
|
||||
|
||||
// Go to sleep
|
||||
Thread->StartRunning.Wait();
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
++ctx->IdleWaitRefCount;
|
||||
Thread->RunningEvents.ThreadSleeping = false;
|
||||
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
static void SleepThread(FEXCore::Context::ContextImpl *CTX, FEXCore::Core::CpuStateFrame *Frame) {
|
||||
CTX->SyscallHandler->SleepThread(CTX, Frame);
|
||||
}
|
||||
|
||||
uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
using ClassPtrType = void (FEXCore::Context::ContextImpl::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
PtrCast CompileBlockPtr;
|
||||
CompileBlockPtr.ClassPtr = &FEXCore::Context::ContextImpl::CompileBlockJit;
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
: Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx}
|
||||
, config {config} {
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx} {
|
||||
EmitDispatcher();
|
||||
}
|
||||
|
||||
Dispatcher::~Dispatcher() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Dispatcher::EmitDispatcher() {
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
@@ -78,8 +60,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
// }
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::ForwardLabel l_Sleep;
|
||||
ARMEmitter::ForwardLabel l_CompileBlock;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -96,9 +78,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
FillStaticRegs();
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
@@ -110,87 +90,86 @@ void Dispatcher::EmitDispatcher() {
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify x2 since it contains our RIP once the block doesn't exist
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
|
||||
auto RipReg = ARMEmitter::XReg::x2;
|
||||
auto RipReg = TMP3;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// L1 Cache
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL , 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
|
||||
br(ARMEmitter::Reg::r3);
|
||||
br(TMP4);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
Bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), VirtualMemorySize - 1);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
}
|
||||
else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 12);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
|
||||
// Load the pointer from the offset
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &NoBlock);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 0x0FFF);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
cmp(ARMEmitter::XReg::x1, RipReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NoBlock);
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x2, ARMEmitter::Reg::r0);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP3, TMP1);
|
||||
|
||||
// Jump to the block
|
||||
br(ARMEmitter::Reg::r3);
|
||||
br(TMP4);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -203,8 +182,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
@@ -221,26 +199,32 @@ void Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP2, 0);
|
||||
|
||||
br(ARMEmitter::Reg::r0);
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, TMP3);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
@@ -248,22 +232,22 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x3, &l_CompileBlock);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
|
||||
// X2 contains our guest RIP
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void *, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
@@ -293,8 +277,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
@@ -304,8 +287,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
brk(0);
|
||||
}
|
||||
@@ -315,8 +297,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
@@ -336,8 +317,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
@@ -393,7 +373,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
|
||||
str(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
@@ -404,112 +384,59 @@ void Dispatcher::EmitDispatcher() {
|
||||
str(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// load static regs
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
FillStaticRegs();
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x0, TMP1);
|
||||
mov(ARMEmitter::XReg::x1, TMP2);
|
||||
mov(ARMEmitter::XReg::x2, TMP3);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, R, Offset);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
// Result is now in x0
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
return Address;
|
||||
};
|
||||
|
||||
{
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
LUREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
LREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
Bind(&l_CTX);
|
||||
dc64(reinterpret_cast<uintptr_t>(CTX));
|
||||
Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
dc64(GetCompileBlockPtr());
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMF.GetConvertedPointer());
|
||||
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
@@ -528,7 +455,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
|
||||
DisasmDecoder->Decode(PCToDecode);
|
||||
auto Output = Disasm.GetOutput();
|
||||
auto Output = Disasm->GetOutput();
|
||||
LogMan::Msg::IFmt("{}", Output);
|
||||
}
|
||||
}
|
||||
@@ -549,40 +476,6 @@ void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_
|
||||
|
||||
#endif
|
||||
|
||||
size_t Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxGDBPauseCheckSize};
|
||||
|
||||
ARMEmitter::ForwardLabel RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::ContextImpl::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Thread));
|
||||
emit.ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, offsetof(FEXCore::Core::InternalThreadState, CTX)); // Get Context
|
||||
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::ContextImpl, Config.RunningMode));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, &RunBlock);
|
||||
{
|
||||
ARMEmitter::ForwardLabel l_GuestRIP;
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(ARMEmitter::XReg::x0, &l_GuestRIP);
|
||||
emit.str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(ARMEmitter::Reg::r0);
|
||||
emit.Bind(&l_GuestRIP);
|
||||
emit.dc64(GuestRIP);
|
||||
}
|
||||
emit.Bind(&RunBlock);
|
||||
|
||||
auto UsedBytes = emit.GetCursorOffset();
|
||||
emit.ClearICache(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
@@ -607,8 +500,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread)
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return fextl::make_unique<Dispatcher>(CTX, Config);
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX) {
|
||||
return fextl::make_unique<Dispatcher>(CTX);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -31,19 +31,15 @@ class ContextImpl;
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
struct DispatcherConfig {
|
||||
bool StaticRegisterAllocation = false;
|
||||
};
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
|
||||
|
||||
class Dispatcher final : public Arm64Emitter {
|
||||
public:
|
||||
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config);
|
||||
~Dispatcher() = default;
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx);
|
||||
~Dispatcher();
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
@@ -71,11 +67,6 @@ public:
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) ;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
@@ -90,7 +81,9 @@ public:
|
||||
#endif
|
||||
|
||||
uint16_t GetSRAGPRCount() const {
|
||||
return StaticRegisters.size();
|
||||
// PF/AF are the final two SRA registers.
|
||||
// Only return the SRA for GPRs.
|
||||
return StaticRegisters.size() - 2;
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const {
|
||||
@@ -98,7 +91,7 @@ public:
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
@@ -109,15 +102,8 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
const DispatcherConfig& GetConfig() const { return config; }
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
DispatcherConfig config;
|
||||
|
||||
static void SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
|
||||
static uint64_t GetCompileBlockPtr();
|
||||
|
||||
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
@@ -284,7 +284,6 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
if (DisplacementSize == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Displacement = DisplacementSize;
|
||||
|
||||
Operand->Type = DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
@@ -1095,7 +1094,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
@@ -1133,6 +1132,10 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1195,9 +1198,9 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
CanContinue = true;
|
||||
}
|
||||
|
||||
bool FinalInstruction = DecodedSize >= CTX->Config.MaxInstPerBlock ||
|
||||
bool FinalInstruction = DecodedSize >= MaxInst ||
|
||||
DecodedSize >= DefaultDecodedBufferSize ||
|
||||
TotalInstructions >= CTX->Config.MaxInstPerBlock;
|
||||
TotalInstructions >= MaxInst;
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
|
||||
@@ -34,7 +34,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::ContextImpl *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
DecodedBlockInformation const *GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
|
||||
@@ -9,14 +9,6 @@
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#define XBYAK64
|
||||
#define XBYAK_CUSTOM_ALLOC
|
||||
#define XBYAK_CUSTOM_MALLOC FEXCore::Allocator::malloc
|
||||
#define XBYAK_CUSTOM_FREE FEXCore::Allocator::free
|
||||
#define XBYAK_CUSTOM_SETS
|
||||
#define XBYAK_STD_UNORDERED_SET fextl::unordered_set
|
||||
#define XBYAK_STD_UNORDERED_MAP fextl::unordered_map
|
||||
#define XBYAK_STD_UNORDERED_MULTIMAP fextl::unordered_multimap
|
||||
#define XBYAK_STD_LIST fextl::list
|
||||
#define XBYAK_NO_EXCEPTION
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
@@ -69,151 +61,63 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
return;
|
||||
}
|
||||
|
||||
const bool DisableAVX = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAVX;
|
||||
const bool EnableAVX = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAVX;
|
||||
LogMan::Throw::AFmt(!(DisableAVX && EnableAVX), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features->FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features->FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
const bool DisableAVX2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAVX2;
|
||||
const bool EnableAVX2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAVX2;
|
||||
LogMan::Throw::AFmt(!(DisableAVX2 && EnableAVX2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#define GET_SINGLE_OPTION(name, enum_name) \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
const bool DisableSVE = HostFeatures() & FEXCore::Config::HostFeatures::DISABLESVE;
|
||||
const bool EnableSVE = HostFeatures() & FEXCore::Config::HostFeatures::ENABLESVE;
|
||||
LogMan::Throw::AFmt(!(DisableSVE && EnableSVE), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX2, AVX2, AVX2);
|
||||
ENABLE_DISABLE_OPTION(SupportsSVE, SVE, SVE);
|
||||
ENABLE_DISABLE_OPTION(SupportsAFP, AFP, AFP);
|
||||
ENABLE_DISABLE_OPTION(SupportsRCPC, LRCPC, LRCPC);
|
||||
ENABLE_DISABLE_OPTION(SupportsTSOImm9, LRCPC2, LRCPC2);
|
||||
ENABLE_DISABLE_OPTION(SupportsCSSC, CSSC, CSSC);
|
||||
ENABLE_DISABLE_OPTION(SupportsPMULL_128Bit, PMULL128, PMULL128);
|
||||
ENABLE_DISABLE_OPTION(SupportsRAND, RNG, RNG);
|
||||
ENABLE_DISABLE_OPTION(SupportsCLZERO, CLZERO, CLZERO);
|
||||
ENABLE_DISABLE_OPTION(SupportsAtomics, Atomics, ATOMICS);
|
||||
ENABLE_DISABLE_OPTION(SupportsFCMA, FCMA, FCMA);
|
||||
ENABLE_DISABLE_OPTION(SupportsFlagM, FlagM, FLAGM);
|
||||
ENABLE_DISABLE_OPTION(SupportsFlagM2, FlagM2, FLAGM2);
|
||||
ENABLE_DISABLE_OPTION(SupportsRPRES, RPRES, RPRES);
|
||||
ENABLE_DISABLE_OPTION(SupportsPreserveAllABI, PRESERVEALLABI, PRESERVEALLABI);
|
||||
GET_SINGLE_OPTION(Crypto, CRYPTO);
|
||||
|
||||
const bool DisableAFP = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAFP;
|
||||
const bool EnableAFP = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAFP;
|
||||
LogMan::Throw::AFmt(!(DisableAFP && EnableAFP), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
#undef GET_SINGLE_OPTION
|
||||
|
||||
const bool DisableLRCPC = HostFeatures() & FEXCore::Config::HostFeatures::DISABLELRCPC;
|
||||
const bool EnableLRCPC = HostFeatures() & FEXCore::Config::HostFeatures::ENABLELRCPC;
|
||||
LogMan::Throw::AFmt(!(DisableLRCPC && EnableLRCPC), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableLRCPC2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLELRCPC2;
|
||||
const bool EnableLRCPC2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLELRCPC2;
|
||||
LogMan::Throw::AFmt(!(DisableLRCPC2 && EnableLRCPC2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableCSSC = HostFeatures() & FEXCore::Config::HostFeatures::DISABLECSSC;
|
||||
const bool EnableCSSC = HostFeatures() & FEXCore::Config::HostFeatures::ENABLECSSC;
|
||||
LogMan::Throw::AFmt(!(DisableCSSC && EnableCSSC), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisablePMULL128 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEPMULL128;
|
||||
const bool EnablePMULL128 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEPMULL128;
|
||||
LogMan::Throw::AFmt(!(DisablePMULL128 && EnablePMULL128), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableRNG = HostFeatures() & FEXCore::Config::HostFeatures::DISABLERNG;
|
||||
const bool EnableRNG = HostFeatures() & FEXCore::Config::HostFeatures::ENABLERNG;
|
||||
LogMan::Throw::AFmt(!(DisableRNG && EnableRNG), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableCLZERO = HostFeatures() & FEXCore::Config::HostFeatures::DISABLECLZERO;
|
||||
const bool EnableCLZERO = HostFeatures() & FEXCore::Config::HostFeatures::ENABLECLZERO;
|
||||
LogMan::Throw::AFmt(!(DisableCLZERO && EnableCLZERO), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableAtomics = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEATOMICS;
|
||||
const bool EnableAtomics = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEATOMICS;
|
||||
LogMan::Throw::AFmt(!(DisableAtomics && EnableAtomics), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFCMA = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFCMA;
|
||||
const bool EnableFCMA = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFCMA;
|
||||
LogMan::Throw::AFmt(!(DisableFCMA && EnableFCMA), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFlagM = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFLAGM;
|
||||
const bool EnableFlagM = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFLAGM;
|
||||
LogMan::Throw::AFmt(!(DisableFlagM && EnableFlagM), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFlagM2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFLAGM2;
|
||||
const bool EnableFlagM2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFLAGM2;
|
||||
LogMan::Throw::AFmt(!(DisableFlagM2 && EnableFlagM2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
if (EnableAVX) {
|
||||
Features->SupportsAVX = true;
|
||||
}
|
||||
else if (DisableAVX) {
|
||||
Features->SupportsAVX = false;
|
||||
}
|
||||
if (EnableAVX2) {
|
||||
Features->SupportsAVX2 = true;
|
||||
}
|
||||
else if (DisableAVX2) {
|
||||
Features->SupportsAVX2 = false;
|
||||
}
|
||||
if (EnableSVE) {
|
||||
Features->SupportsSVE = true;
|
||||
}
|
||||
else if (DisableSVE) {
|
||||
Features->SupportsSVE = false;
|
||||
}
|
||||
if (EnableAFP) {
|
||||
Features->SupportsFlushInputsToZero = true;
|
||||
}
|
||||
else if (DisableAFP) {
|
||||
Features->SupportsFlushInputsToZero = false;
|
||||
}
|
||||
if (EnableLRCPC) {
|
||||
Features->SupportsRCPC = true;
|
||||
}
|
||||
else if (DisableLRCPC) {
|
||||
Features->SupportsRCPC = false;
|
||||
}
|
||||
if (EnableLRCPC2) {
|
||||
Features->SupportsTSOImm9 = true;
|
||||
}
|
||||
else if (DisableLRCPC2) {
|
||||
Features->SupportsTSOImm9 = false;
|
||||
}
|
||||
if (EnableCSSC) {
|
||||
Features->SupportsCSSC = true;
|
||||
}
|
||||
else if (DisableCSSC) {
|
||||
Features->SupportsCSSC = false;
|
||||
}
|
||||
if (EnablePMULL128) {
|
||||
if (EnableCrypto) {
|
||||
Features->SupportsAES = true;
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsSHA = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
}
|
||||
else if (DisablePMULL128) {
|
||||
else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsSHA = false;
|
||||
Features->SupportsPMULL_128Bit = false;
|
||||
}
|
||||
if (EnableRNG) {
|
||||
Features->SupportsRAND = true;
|
||||
}
|
||||
else if (DisableRNG) {
|
||||
Features->SupportsRAND = false;
|
||||
}
|
||||
if (EnableCLZERO) {
|
||||
Features->SupportsCLZERO = true;
|
||||
}
|
||||
else if (DisableCLZERO) {
|
||||
Features->SupportsCLZERO = false;
|
||||
}
|
||||
if (EnableAtomics) {
|
||||
Features->SupportsAtomics = true;
|
||||
}
|
||||
else if (DisableAtomics) {
|
||||
Features->SupportsAtomics = false;
|
||||
}
|
||||
if (EnableFCMA) {
|
||||
Features->SupportsFCMA = true;
|
||||
}
|
||||
else if (DisableFCMA) {
|
||||
Features->SupportsFCMA = false;
|
||||
}
|
||||
if (EnableFlagM) {
|
||||
Features->SupportsFlagM = true;
|
||||
}
|
||||
else if (DisableFlagM) {
|
||||
Features->SupportsFlagM = false;
|
||||
}
|
||||
if (EnableFlagM2) {
|
||||
Features->SupportsFlagM2 = true;
|
||||
}
|
||||
else if (DisableFlagM2) {
|
||||
Features->SupportsFlagM2 = false;
|
||||
}
|
||||
}
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
// Vixl simulator doesn't support AFP.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kAFP);
|
||||
// Vixl simulator doesn't support RPRES.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kRPRES);
|
||||
#elif !defined(_WIN32)
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
@@ -223,11 +127,13 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) &&
|
||||
Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsAFP = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
@@ -235,6 +141,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsFCMA = Features.Has(vixl::CPUFeatures::Feature::kFcma);
|
||||
SupportsFlagM = Features.Has(vixl::CPUFeatures::Feature::kFlagM);
|
||||
SupportsFlagM2 = Features.Has(vixl::CPUFeatures::Feature::kAXFlag);
|
||||
SupportsRPRES = Features.Has(vixl::CPUFeatures::Feature::kRPRES);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
@@ -249,11 +156,15 @@ HostFeatures::HostFeatures() {
|
||||
#endif
|
||||
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
|
||||
SupportsAVX2 = false;
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
|
||||
SupportsAFP = false;
|
||||
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
|
||||
SupportsRPRES = false;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
@@ -291,6 +202,8 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
// Simulator doesn't support SHA
|
||||
SupportsSHA = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
@@ -336,10 +249,11 @@ HostFeatures::HostFeatures() {
|
||||
SupportsCLZERO = data[1] & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsAFP = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#endif
|
||||
#endif
|
||||
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
|
||||
OverrideFeatures(this);
|
||||
}
|
||||
}
|
||||
@@ -1,2 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
@@ -85,7 +85,7 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle);
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80CVTTO: {
|
||||
@@ -93,11 +93,11 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -107,11 +107,11 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -124,28 +124,28 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, SupportsPreserveAllABI};
|
||||
}
|
||||
else {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, SupportsPreserveAllABI};
|
||||
}
|
||||
else {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, SupportsPreserveAllABI};
|
||||
}
|
||||
else {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -167,7 +167,7 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -176,11 +176,11 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -190,13 +190,13 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
@@ -244,10 +244,10 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
|
||||
// SSE4.2 Fallbacks
|
||||
case IR::OP_VPCMPESTRX:
|
||||
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
|
||||
default:
|
||||
|
||||
@@ -45,6 +45,6 @@ namespace FEXCore::CPU {
|
||||
class InterpreterOps {
|
||||
public:
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
|
||||
static bool GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_Header const *IROp, FallbackInfo *Info);
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -85,91 +85,149 @@ DEF_OP(Add) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AddNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
DEF_OP(AddWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AddWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
cmn(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
adds(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
cmn(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
adds(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::NZCV);
|
||||
DEF_OP(AddShift) {
|
||||
auto Op = IROp->C<IR::IROp_AddShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
add(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(AddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AddNZCV>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
|
||||
cmn(EmitSize, Src1, Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
|
||||
|
||||
if (OpSize < 4) {
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmn(EmitSize, Src1, GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AdcNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AdcNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
// TODO: Optimize this out
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->NZCV.ID()));
|
||||
|
||||
adcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
DEF_OP(AdcWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AdcWithFlags>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
adcs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Adc) {
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
adc(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(SbbWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_SbbWithFlags>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sbcs(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(SbbNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SbbNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
// Carry-in needs to be inverted for subtractions due to carry versus borrow
|
||||
// distinction between x86 and arm.
|
||||
// See below remarks on cfinv
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, GetReg(Op->NZCV.ID()), 1u << 29);
|
||||
|
||||
// TODO: Optimize this out
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
sbcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
DEF_OP(Sbb) {
|
||||
auto Op = IROp->C<IR::IROp_Sbb>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
// The carry flag produced by arm64 sbcs is inverted compared to the x86 carry
|
||||
// flag. Invert it now.
|
||||
//
|
||||
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
|
||||
// that's only available with Feat_FlagM. For now the portable way is to flip
|
||||
// bit 29 (carry) manually.
|
||||
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sbc(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src1.ID());
|
||||
uint64_t Const;
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// setf+rmif would avoid the scratch register, but higher latency on M1.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
// zeroing CV.
|
||||
if (OpSize < 4) {
|
||||
lsl(EmitSize, Dst, Src, 32 - (OpSize * 8));
|
||||
Src = Dst;
|
||||
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
|
||||
// needed.
|
||||
if (Op->Src1 != Op->Src2) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, TMP1, Src1, Const);
|
||||
} else {
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
}
|
||||
|
||||
Src1 = TMP1;
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
tst(EmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
@@ -183,40 +241,138 @@ DEF_OP(Sub) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
sub(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(SubWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_SubWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
subs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), Const);
|
||||
} else {
|
||||
subs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
} else if (IsInlineConstant(Op->Src1, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
|
||||
cmp(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src2.ID()));
|
||||
} else {
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
|
||||
ARMEmitter::Register ShiftedSrc1 = GetZeroableReg(Op->Src1);
|
||||
|
||||
// Shift to fix flags for <32-bit ops.
|
||||
// Any shift of zero is still zero so optimize out silly zero shifts.
|
||||
if (OpSize < 4 && ShiftedSrc1 != ARMEmitter::Reg::zr) {
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, ShiftedSrc1, Shift);
|
||||
ShiftedSrc1 = TMP1;
|
||||
}
|
||||
|
||||
if (OpSize < 4) {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
DEF_OP(CarryInvert) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
cfinv();
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
DEF_OP(RmifNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_RmifNZCV>();
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
if (Op->InvertCarry) {
|
||||
// The carry flag produced by arm64 subs is inverted compared to the x86 carry
|
||||
// flag. Invert it now.
|
||||
//
|
||||
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
|
||||
// that's only available with Feat_FlagM. For now the portable way is to flip
|
||||
// bit 29 (carry) manually.
|
||||
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
|
||||
rmif(GetReg(Op->Src.ID()).X(), Op->Rotate, Op->Mask);
|
||||
}
|
||||
|
||||
DEF_OP(SetSmallNZV) {
|
||||
auto Op = IROp->C<IR::IROp_SetSmallNZV>();
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 1 || OpSize == 2, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
if (OpSize == 1) {
|
||||
setf8(GetReg(Op->Src.ID()).W());
|
||||
} else {
|
||||
setf16(GetReg(Op->Src.ID()).W());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AXFlag) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
|
||||
axflag();
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
|
||||
uint64_t Const = 0;
|
||||
auto Src1 = GetZeroableReg(Op->Src1);
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmn(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
|
||||
} else {
|
||||
ccmn(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -227,27 +383,10 @@ DEF_OP(Neg) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
|
||||
abs(EmitSize, Dst, Src);
|
||||
}
|
||||
else {
|
||||
cmp(EmitSize, Src, 0);
|
||||
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
if (Op->Cond == FEXCore::IR::COND_AL)
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
else
|
||||
cneg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()), MapSelectCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
@@ -270,6 +409,16 @@ DEF_OP(UMul) {
|
||||
mul(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(UMull) {
|
||||
auto Op = IROp->C<IR::IROp_UMull>();
|
||||
umull(GetReg(Node).X(), GetReg(Op->Src1.ID()).W(), GetReg(Op->Src2.ID()).W());
|
||||
}
|
||||
|
||||
DEF_OP(SMull) {
|
||||
auto Op = IROp->C<IR::IROp_SMull>();
|
||||
smull(GetReg(Node).X(), GetReg(Op->Src1.ID()).W(), GetReg(Op->Src2.ID()).W());
|
||||
}
|
||||
|
||||
DEF_OP(Div) {
|
||||
auto Op = IROp->C<IR::IROp_Div>();
|
||||
|
||||
@@ -515,6 +664,41 @@ DEF_OP(And) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AndWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AndWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
// See TestNZ
|
||||
if (OpSize < 4) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (Src1 != Src2) {
|
||||
and_(EmitSize, Dst, Src1, Src2);
|
||||
} else if (Dst != Src1) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, Src1);
|
||||
}
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Dst, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ands(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
ands(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -549,6 +733,16 @@ DEF_OP(Xor) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -655,64 +849,58 @@ DEF_OP(PDep) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Input = GetReg(Op->Input.ID());
|
||||
const auto Mask = GetReg(Op->Mask.ID());
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
const auto ShiftedBitReg = TMP1.R();
|
||||
const auto BitReg = TMP2.R();
|
||||
const auto SubMaskReg = TMP3.R();
|
||||
const auto IndexReg = TMP4.R();
|
||||
const auto ZeroReg = ARMEmitter::Reg::zr;
|
||||
// PDep implementation follows the ideas from
|
||||
// http://0x80.pl/articles/pdep-soft-emu.html ... Basically, iterate the *set*
|
||||
// bits only, which will be faster than the naive implementation as long as
|
||||
// there are enough holes in the mask.
|
||||
//
|
||||
// The specific arm64 assembly used is based on the sequence that clang
|
||||
// generates for the C code, giving context to the scheduling yielding better
|
||||
// ILP than I would do by hand. The registers are allocated by hand however,
|
||||
// to fit within the tight constraints we have here withot spilling. Also, we
|
||||
// use cbz/cbnz for conditional branching to avoid clobbering NZCV.
|
||||
|
||||
const auto InputReg = StaticRegisters[0];
|
||||
const auto MaskReg = StaticRegisters[1];
|
||||
const auto DestReg = StaticRegisters[2];
|
||||
// We can't clobber these
|
||||
const auto OrigInput = GetReg(Op->Input.ID());
|
||||
const auto OrigMask = GetReg(Op->Mask.ID());
|
||||
|
||||
const auto SpillCode = 1U << InputReg.Idx() |
|
||||
1U << MaskReg.Idx() |
|
||||
1U << DestReg.Idx();
|
||||
// So we have shadow as temporaries
|
||||
const auto Input = TMP1.R();
|
||||
const auto Mask = TMP2.R();
|
||||
|
||||
// these get used variously as scratch
|
||||
const auto T0 = TMP3.R();
|
||||
const auto T1 = TMP4.R();
|
||||
|
||||
ARMEmitter::ForwardLabel EarlyExit;
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, IndexReg, ZeroReg);
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(TMP1, false, SpillCode);
|
||||
// First, copy the input/mask, since we'll be clobbering. Copy as 64-bit to
|
||||
// make this 0-uop on Firestorm.
|
||||
mov(ARMEmitter::Size::i64Bit, Input, OrigInput);
|
||||
mov(ARMEmitter::Size::i64Bit, Mask, OrigMask);
|
||||
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
cbz(EmitSize, OrigMask, &Done);
|
||||
|
||||
mov(EmitSize, InputReg, Input);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
mov(EmitSize, DestReg, ZeroReg);
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
and_(EmitSize, T0, T0, Mask);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
rbit(EmitSize, ShiftedBitReg, MaskReg);
|
||||
clz(EmitSize, ShiftedBitReg, ShiftedBitReg);
|
||||
lsrv(EmitSize, BitReg, InputReg, IndexReg);
|
||||
and_(EmitSize, BitReg, BitReg, 1);
|
||||
sub(EmitSize, SubMaskReg, MaskReg, 1);
|
||||
add(EmitSize, IndexReg, IndexReg, 1);
|
||||
ands(EmitSize, MaskReg, MaskReg, SubMaskReg);
|
||||
lslv(EmitSize, ShiftedBitReg, BitReg, ShiftedBitReg);
|
||||
orr(EmitSize, DestReg, DestReg, ShiftedBitReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NextBit);
|
||||
// Store result in a temp so it doesn't get clobbered.
|
||||
// and restore it after the re-fill below.
|
||||
mov(EmitSize, IndexReg, DestReg);
|
||||
// Restore our registers before leaving
|
||||
// TODO: Also remove along with above TODO.
|
||||
FillStaticRegs(false, SpillCode);
|
||||
mov(EmitSize, Dest, IndexReg);
|
||||
b(&Done);
|
||||
|
||||
// Early exit
|
||||
Bind(&EarlyExit);
|
||||
mov(EmitSize, Dest, ZeroReg);
|
||||
sbfx(EmitSize, T1, Input, 0, 1);
|
||||
eor(EmitSize, Mask, Mask, T0);
|
||||
and_(EmitSize, T0, T1, T0);
|
||||
neg(EmitSize, T1, Mask);
|
||||
orr(EmitSize, Dest, Dest, T0);
|
||||
lsr(EmitSize, Input, Input, 1);
|
||||
and_(EmitSize, T0, Mask, T1);
|
||||
cbnz(EmitSize, T0, &NextBit);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
@@ -734,9 +922,9 @@ DEF_OP(PExt) {
|
||||
const auto BitReg = TMP2;
|
||||
const auto ValueReg = TMP3;
|
||||
|
||||
ARMEmitter::ForwardLabel EarlyExit;
|
||||
ARMEmitter::SingleUseForwardLabel EarlyExit;
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
@@ -790,8 +978,8 @@ DEF_OP(LDiv) {
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ARMEmitter::ForwardLabel Only64Bit{};
|
||||
ARMEmitter::ForwardLabel LongDIVRet{};
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit{};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
|
||||
|
||||
// Check if the upper bits match the top bit of the lower 64-bits
|
||||
// Sign extend the top bit of lower bits
|
||||
@@ -803,18 +991,18 @@ DEF_OP(LDiv) {
|
||||
|
||||
// Long divide
|
||||
{
|
||||
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
|
||||
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
mov(EmitSize, TMP1, Upper);
|
||||
mov(EmitSize, TMP2, Lower);
|
||||
mov(EmitSize, TMP3, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(TMP4);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
mov(EmitSize, Dst, TMP1);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
@@ -862,8 +1050,8 @@ DEF_OP(LUDiv) {
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ARMEmitter::ForwardLabel Only64Bit{};
|
||||
ARMEmitter::ForwardLabel LongDIVRet{};
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit{};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
@@ -871,18 +1059,18 @@ DEF_OP(LUDiv) {
|
||||
|
||||
// Long divide
|
||||
{
|
||||
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
|
||||
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
mov(EmitSize, TMP1, Upper);
|
||||
mov(EmitSize, TMP2, Lower);
|
||||
mov(EmitSize, TMP3, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(TMP4);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
mov(EmitSize, Dst, TMP1);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
@@ -934,8 +1122,8 @@ DEF_OP(LRem) {
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ARMEmitter::ForwardLabel Only64Bit{};
|
||||
ARMEmitter::ForwardLabel LongDIVRet{};
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit{};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
|
||||
|
||||
// Check if the upper bits match the top bit of the lower 64-bits
|
||||
// Sign extend the top bit of lower bits
|
||||
@@ -947,18 +1135,18 @@ DEF_OP(LRem) {
|
||||
|
||||
// Long divide
|
||||
{
|
||||
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
|
||||
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
mov(EmitSize, TMP1, Upper);
|
||||
mov(EmitSize, TMP2, Lower);
|
||||
mov(EmitSize, TMP3, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(TMP4);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
mov(EmitSize, Dst, TMP1);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
@@ -1008,8 +1196,8 @@ DEF_OP(LURem) {
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ARMEmitter::ForwardLabel Only64Bit{};
|
||||
ARMEmitter::ForwardLabel LongDIVRet{};
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit{};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
@@ -1017,18 +1205,18 @@ DEF_OP(LURem) {
|
||||
|
||||
// Long divide
|
||||
{
|
||||
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
|
||||
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
mov(EmitSize, TMP1, Upper);
|
||||
mov(EmitSize, TMP2, Lower);
|
||||
mov(EmitSize, TMP3, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(TMP4);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
mov(EmitSize, Dst, TMP1);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
@@ -1161,6 +1349,11 @@ DEF_OP(FindTrailingZeroes) {
|
||||
rbit(EmitSize, Dst, Src);
|
||||
|
||||
if (OpSize == 2) {
|
||||
// This orr does two things. First, if the (masked) source is zero, it
|
||||
// reverses to zero in the top so it forces clz to return 16. Second, it
|
||||
// ensures garbage in the upper bits of the source don't affect clz, because
|
||||
// they'll rbit to garbage in the bottom below the 0x8000 and be ignored by
|
||||
// the clz. So we handle Src upper garbage without explicitly masking.
|
||||
orr(EmitSize, Dst, Dst, 0x8000);
|
||||
}
|
||||
|
||||
@@ -1178,6 +1371,8 @@ DEF_OP(CountLeadingZeroes) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (OpSize == 2) {
|
||||
// Expressing as lsl+orr+clz clears away any garbage in the upper bits
|
||||
// (alternatively could do uxth+clz+sub.. equal cost in total).
|
||||
lsl(EmitSize, Dst, Src, 16);
|
||||
orr(EmitSize, Dst, Dst, 0x8000);
|
||||
clz(EmitSize, Dst, Dst);
|
||||
@@ -1217,6 +1412,11 @@ DEF_OP(Bfi) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
@@ -1246,6 +1446,11 @@ DEF_OP(Bfxil) {
|
||||
// If Dst and SrcDst match then this turns in to a single instruction.
|
||||
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
@@ -1287,36 +1492,6 @@ DEF_OP(Sbfe) {
|
||||
sbfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1325,28 +1500,15 @@ DEF_OP(Select) {
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
|
||||
if (tests) {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
tst(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
tst(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
@@ -1385,6 +1547,35 @@ DEF_OP(Select) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelect) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
if (is_const_true) {
|
||||
if (is_const_false != true || !(const_true == 1 || const_true == all_ones) || const_false != 0) {
|
||||
LOGMAN_MSG_A_FMT("NZCVSelect: Unsupported constant");
|
||||
}
|
||||
|
||||
if (const_true == all_ones)
|
||||
csetm(EmitSize, Dst, cc);
|
||||
else
|
||||
cset(EmitSize, Dst, cc);
|
||||
} else {
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), cc);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1499,41 +1690,10 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
const auto EmitSubSize = Op->ElementSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
|
||||
|
||||
fcmp(EmitSubSize, Scalar1, Scalar2);
|
||||
bool set = false;
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
|
||||
LOGMAN_THROW_AA_FMT(IR::FCMP_FLAG_EQ == 0, "IR::FCMP_FLAG_EQ must equal 0");
|
||||
// EQ or unordered
|
||||
cset(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Condition::CC_EQ); // Z = 1
|
||||
csinc(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_VC); // IF !V ? Z : 1
|
||||
set = true;
|
||||
}
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
// LT or unordered
|
||||
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_LT);
|
||||
if (!set) {
|
||||
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT);
|
||||
set = true;
|
||||
} else {
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT, 1);
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
|
||||
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_VS);
|
||||
if (!set) {
|
||||
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED);
|
||||
set = true;
|
||||
} else {
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -32,8 +32,8 @@ DEF_OP(CASPair) {
|
||||
}
|
||||
else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
@@ -82,8 +82,8 @@ DEF_OP(CAS) {
|
||||
}
|
||||
else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (OpSize == 1) {
|
||||
@@ -193,6 +193,33 @@ DEF_OP(AtomicAnd) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicCLR) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicCLR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
bic(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -247,6 +274,27 @@ DEF_OP(AtomicXor) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -262,8 +310,7 @@ DEF_OP(AtomicSwap) {
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP2, Src);
|
||||
ldswpal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
|
||||
@@ -11,9 +11,9 @@ $end_info$
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
@@ -53,32 +53,30 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
ARMEmitter::ForwardLabel l_BranchHost;
|
||||
ARMEmitter::ForwardLabel l_BranchGuest;
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
ldr(TMP1, &l_BranchHost);
|
||||
blr(TMP1);
|
||||
|
||||
Bind(&l_BranchHost);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
Bind(&l_BranchGuest);
|
||||
dc64(NewRIP);
|
||||
|
||||
} else {
|
||||
|
||||
ARMEmitter::ForwardLabel FullLookup;
|
||||
ARMEmitter::SingleUseForwardLabel FullLookup;
|
||||
auto RipReg = GetReg(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::XReg::x0, RipReg.X());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
br(ARMEmitter::Reg::r1);
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
br(TMP2);
|
||||
|
||||
Bind(&FullLookup);
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
@@ -96,9 +94,7 @@ DEF_OP(Jump) {
|
||||
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
@@ -116,8 +112,8 @@ static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
@@ -129,42 +125,27 @@ DEF_OP(CondJump) {
|
||||
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
if (Op->FromNZCV) {
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (tests) {
|
||||
if (isConst) {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
} else {
|
||||
if (isConst) {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ ||
|
||||
Op->Cond.Val == FEXCore::IR::COND_NEQ,
|
||||
"CondJump: Expected simple condition");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
}
|
||||
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
// TODO: Wire up tbz/tbnz
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
@@ -369,58 +350,58 @@ DEF_OP(ValidateCode) {
|
||||
int idx = 0;
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry + Op->Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 1);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
while (len >= 8)
|
||||
{
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
ldr(ARMEmitter::XReg::x2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 8;
|
||||
idx += 8;
|
||||
}
|
||||
while (len >= 4)
|
||||
{
|
||||
ldr(ARMEmitter::WReg::w2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
ldr(ARMEmitter::WReg::w2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 4;
|
||||
idx += 4;
|
||||
}
|
||||
while (len >= 2)
|
||||
{
|
||||
ldrh(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint16_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
ldrh(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 2;
|
||||
idx += 2;
|
||||
}
|
||||
while (len >= 1)
|
||||
{
|
||||
ldrb(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint8_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
|
||||
ldrb(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t *)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 1;
|
||||
idx += 1;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
@@ -439,16 +420,23 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, GetReg(Op->Leaf.ID()));
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
@@ -456,6 +444,11 @@ DEF_OP(CPUID) {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -463,21 +456,22 @@ DEF_OP(CPUID) {
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
DEF_OP(XGetBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
@@ -485,6 +479,10 @@ DEF_OP(XGETBV) {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -492,8 +490,8 @@ DEF_OP(XGETBV) {
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -12,12 +12,12 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
DEF_OP(VAESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -44,7 +44,7 @@ DEF_OP(AESEnc) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -69,7 +69,7 @@ DEF_OP(AESEncLast) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -96,7 +96,7 @@ DEF_OP(AESDec) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -121,7 +121,7 @@ DEF_OP(AESDecLast) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
DEF_OP(VAESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
@@ -179,6 +179,32 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256su0(Dst, Src2);
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256su0(VTMP1, Src2);
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -18,17 +18,18 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -78,11 +79,45 @@ namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
if (!InterpreterOps::GetFallbackHandler(CTX->HostFeatures.SupportsPreserveAllABI, IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Op: {}", FEXCore::IR::GetName(IROp->Op));
|
||||
#endif
|
||||
} else {
|
||||
auto FillF80Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
mov(TMP2, ARMEmitter::XReg::x1);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, TMP2);
|
||||
};
|
||||
|
||||
auto FillF64Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.D(), ARMEmitter::DReg::d0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), VTMP1.D());
|
||||
};
|
||||
|
||||
auto FillI32Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1.W(), ARMEmitter::WReg::w0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), TMP1.W());
|
||||
};
|
||||
|
||||
switch(Info.ABI) {
|
||||
case FABI_F80_I16_F32:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
@@ -98,12 +133,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
FillF80Result();
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -121,12 +151,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
FillF80Result();
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -135,13 +160,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
@@ -150,12 +175,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
FillF80Result();
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -176,10 +196,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VTMP1.S(), ARMEmitter::SReg::s0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
fmov(Dst.S(), ARMEmitter::SReg::s0);
|
||||
fmov(Dst.S(), VTMP1.S());
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -200,10 +223,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), ARMEmitter::DReg::d0);
|
||||
FillF64Result();
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -222,21 +242,24 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), ARMEmitter::DReg::d0);
|
||||
FillF64Result();
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
mov(ARMEmitter::DReg::d1, Src2.D());
|
||||
mov(VTMP1.D(), Src1.D());
|
||||
mov(VTMP2.D(), Src2.D());
|
||||
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::DReg::d0, VTMP1.D());
|
||||
mov(ARMEmitter::DReg::d1, VTMP2.D());
|
||||
}
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
@@ -246,10 +269,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.D(), ARMEmitter::DReg::d0);
|
||||
FillF64Result();
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -270,10 +290,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I16_F80:{
|
||||
@@ -293,10 +316,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, ARMEmitter::Reg::r0);
|
||||
FillI32Result();
|
||||
}
|
||||
break;
|
||||
case FABI_I64_I16_F80:{
|
||||
@@ -315,10 +335,14 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_I16_F80_F80:{
|
||||
@@ -341,10 +365,14 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_I16_F80:{
|
||||
@@ -364,12 +392,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
FillF80Result();
|
||||
}
|
||||
break;
|
||||
case FABI_F80_I16_F80_F80:{
|
||||
@@ -393,27 +416,28 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
FillF80Result();
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
mov(TMP1, SrcRAX.X());
|
||||
mov(TMP2, SrcRDX.X());
|
||||
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP3, true);
|
||||
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x0, TMP1);
|
||||
mov(ARMEmitter::XReg::x1, TMP2);
|
||||
}
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src1, 1);
|
||||
@@ -431,12 +455,9 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r7);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), ARMEmitter::WReg::w0);
|
||||
break;
|
||||
FillI32Result();
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
@@ -462,12 +483,9 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), ARMEmitter::WReg::w0);
|
||||
break;
|
||||
FillI32Result();
|
||||
}
|
||||
break;
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -480,9 +498,26 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record) - 8;
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
|
||||
FEXCore::ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
emit.ldr(TMP1, &l_BranchHost);
|
||||
emit.blr(TMP1);
|
||||
emit.Bind(&l_BranchHost);
|
||||
emit.dc64(LinkerAddress);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 8);
|
||||
}
|
||||
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Record->HostBranch = LinkerAddress;
|
||||
}
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
@@ -491,35 +526,24 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record) - 8;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (vixl::IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 4);
|
||||
emit.b(offset);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 4);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
emit.blr(FEXCore::ARMEmitter::Reg::r0);
|
||||
emit.Bind(&l_BranchHost);
|
||||
emit.dc64(LinkerAddress);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, DirectBlockDelinker);
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
Record->HostBranch = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
@@ -530,9 +554,11 @@ void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128{ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256{ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsRPRES{ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP{ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
@@ -572,8 +598,11 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::HLE::SyscallHandler::HandleSyscall);
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
|
||||
}
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
@@ -592,15 +621,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
ClearCache();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (CTX->Dispatcher->GetConfig().StaticRegisterAllocation) {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegisterSRA;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegisterSRA;
|
||||
}
|
||||
else {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegister;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegister;
|
||||
}
|
||||
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
@@ -687,8 +707,7 @@ bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -700,7 +719,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
@@ -744,16 +763,11 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
#ifdef _WIN32
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(CodeData.BlockEntry, Entry);
|
||||
CursorIncrement(GDBSize);
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -766,8 +780,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
if (vixl::aarch64::Assembler::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, TotalSpillSlotsSize);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, ARMEmitter::XReg::x0, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -800,272 +814,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(ADDNZCV, AddNZCV);
|
||||
REGISTER_OP(ADCNZCV, AdcNZCV);
|
||||
REGISTER_OP(SBBNZCV, SbbNZCV);
|
||||
REGISTER_OP(TESTNZ, TestNZ);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(SUBNZCV, SubNZCV);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
REGISTER_OP(UDIV, UDiv);
|
||||
REGISTER_OP(REM, Rem);
|
||||
REGISTER_OP(UREM, URem);
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(ORNROR, Ornror);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
REGISTER_OP(LUREM, LURem);
|
||||
REGISTER_OP(NOT, Not);
|
||||
REGISTER_OP(POPCOUNT, Popcount);
|
||||
REGISTER_OP(FINDLSB, FindLSB);
|
||||
REGISTER_OP(FINDMSB, FindMSB);
|
||||
REGISTER_OP(FINDTRAILINGZEROES, FindTrailingZeroes);
|
||||
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
|
||||
REGISTER_OP(REV, Rev);
|
||||
REGISTER_OP(BFI, Bfi);
|
||||
REGISTER_OP(BFXIL, Bfxil);
|
||||
REGISTER_OP(BFE, Bfe);
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
|
||||
// Atomic ops
|
||||
REGISTER_OP(CASPAIR, CASPair);
|
||||
REGISTER_OP(CAS, CAS);
|
||||
REGISTER_OP(ATOMICADD, AtomicAdd);
|
||||
REGISTER_OP(ATOMICSUB, AtomicSub);
|
||||
REGISTER_OP(ATOMICAND, AtomicAnd);
|
||||
REGISTER_OP(ATOMICOR, AtomicOr);
|
||||
REGISTER_OP(ATOMICXOR, AtomicXor);
|
||||
REGISTER_OP(ATOMICSWAP, AtomicSwap);
|
||||
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
|
||||
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
|
||||
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
|
||||
REGISTER_OP(ATOMICFETCHCLR, AtomicFetchCLR);
|
||||
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
|
||||
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
REGISTER_OP(TELEMETRYSETVALUE, TelemetrySetValue);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(VDUPFROMGPR, VDupFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// Flag ops
|
||||
REGISTER_OP(GETHOSTFLAG, GetHostFlag);
|
||||
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
REGISTER_OP(FILLREGISTER, FillRegister);
|
||||
REGISTER_OP(LOADFLAG, LoadFlag);
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(VLOADVECTORELEMENT, VLoadVectorElement);
|
||||
REGISTER_OP(VSTOREVECTORELEMENT, VStoreVectorElement);
|
||||
REGISTER_OP(VBROADCASTFROMMEM, VBroadcastFromMem);
|
||||
REGISTER_OP(PUSH, Push);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
// Misc ops
|
||||
REGISTER_OP(DUMMY, NoOp);
|
||||
REGISTER_OP(IRHEADER, NoOp);
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PRINT, Print);
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(LOADNAMEDVECTORCONSTANT, LoadNamedVectorConstant);
|
||||
REGISTER_OP(LOADNAMEDVECTORINDEXEDCONSTANT, LoadNamedVectorIndexedConstant);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
REGISTER_OP(VOR, VOr);
|
||||
REGISTER_OP(VXOR, VXor);
|
||||
REGISTER_OP(VADD, VAdd);
|
||||
REGISTER_OP(VSUB, VSub);
|
||||
REGISTER_OP(VUQADD, VUQAdd);
|
||||
REGISTER_OP(VUQSUB, VUQSub);
|
||||
REGISTER_OP(VSQADD, VSQAdd);
|
||||
REGISTER_OP(VSQSUB, VSQSub);
|
||||
REGISTER_OP(VADDP, VAddP);
|
||||
REGISTER_OP(VADDV, VAddV);
|
||||
REGISTER_OP(VUMINV, VUMinV);
|
||||
REGISTER_OP(VURAVG, VURAvg);
|
||||
REGISTER_OP(VABS, VAbs);
|
||||
REGISTER_OP(VFABS, VFAbs);
|
||||
REGISTER_OP(VPOPCOUNT, VPopcount);
|
||||
REGISTER_OP(VFADD, VFAdd);
|
||||
REGISTER_OP(VFADDP, VFAddP);
|
||||
REGISTER_OP(VFSUB, VFSub);
|
||||
REGISTER_OP(VFMUL, VFMul);
|
||||
REGISTER_OP(VFDIV, VFDiv);
|
||||
REGISTER_OP(VFMIN, VFMin);
|
||||
REGISTER_OP(VFMAX, VFMax);
|
||||
REGISTER_OP(VFRECP, VFRecp);
|
||||
REGISTER_OP(VFSQRT, VFSqrt);
|
||||
REGISTER_OP(VFRSQRT, VFRSqrt);
|
||||
REGISTER_OP(VNEG, VNeg);
|
||||
REGISTER_OP(VFNEG, VFNeg);
|
||||
REGISTER_OP(VNOT, VNot);
|
||||
REGISTER_OP(VUMIN, VUMin);
|
||||
REGISTER_OP(VSMIN, VSMin);
|
||||
REGISTER_OP(VUMAX, VUMax);
|
||||
REGISTER_OP(VSMAX, VSMax);
|
||||
REGISTER_OP(VZIP, VZip);
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VTRN, VTrn);
|
||||
REGISTER_OP(VTRN2, VTrn2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
REGISTER_OP(VCMPGT, VCMPGT);
|
||||
REGISTER_OP(VCMPGTZ, VCMPGTZ);
|
||||
REGISTER_OP(VCMPLTZ, VCMPLTZ);
|
||||
REGISTER_OP(VFCMPEQ, VFCMPEQ);
|
||||
REGISTER_OP(VFCMPNEQ, VFCMPNEQ);
|
||||
REGISTER_OP(VFCMPLT, VFCMPLT);
|
||||
REGISTER_OP(VFCMPGT, VFCMPGT);
|
||||
REGISTER_OP(VFCMPLE, VFCMPLE);
|
||||
REGISTER_OP(VFCMPORD, VFCMPORD);
|
||||
REGISTER_OP(VFCMPUNO, VFCMPUNO);
|
||||
REGISTER_OP(VUSHL, VUShl);
|
||||
REGISTER_OP(VUSHR, VUShr);
|
||||
REGISTER_OP(VSSHR, VSShr);
|
||||
REGISTER_OP(VUSHLS, VUShlS);
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VUSHRSWIDE, VUShrSWide);
|
||||
REGISTER_OP(VSSHRSWIDE, VSShrSWide);
|
||||
REGISTER_OP(VUSHLSWIDE, VUShlSWide);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
REGISTER_OP(VUXTL2, VUXTL2);
|
||||
REGISTER_OP(VSQXTN, VSQXTN);
|
||||
REGISTER_OP(VSQXTN2, VSQXTN2);
|
||||
REGISTER_OP(VSQXTNPAIR, VSQXTNPair);
|
||||
REGISTER_OP(VSQXTUN, VSQXTUN);
|
||||
REGISTER_OP(VSQXTUN2, VSQXTUN2);
|
||||
REGISTER_OP(VSQXTUNPAIR, VSQXTUNPair);
|
||||
REGISTER_OP(VSRSHR, VSRSHR);
|
||||
REGISTER_OP(VSQSHL, VSQSHL);
|
||||
REGISTER_OP(VUMUL, VMul);
|
||||
REGISTER_OP(VSMUL, VMul);
|
||||
REGISTER_OP(VUMULL, VUMull);
|
||||
REGISTER_OP(VSMULL, VSMull);
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUMULH, VUMulH);
|
||||
REGISTER_OP(VSMULH, VSMulH);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VTBL2, VTBL2);
|
||||
REGISTER_OP(VREV32, VRev32);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VFCADD, VFCADD);
|
||||
#define IROP_DISPATCH_DISPATCH
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef REGISTER_OP
|
||||
|
||||
default:
|
||||
@@ -1107,6 +858,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
@@ -1148,7 +900,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
LogMan::Msg::IFmt("Disassemble Begin");
|
||||
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
|
||||
DisasmDecoder->Decode(PCToDecode);
|
||||
auto Output = Disasm.GetOutput();
|
||||
auto Output = Disasm->GetOutput();
|
||||
LogMan::Msg::IFmt("{}", Output);
|
||||
}
|
||||
LogMan::Msg::IFmt("Disassemble End");
|
||||
@@ -1176,8 +928,8 @@ void Arm64JITCore::ResetStack() {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, TotalSpillSlotsSize);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, ARMEmitter::XReg::x0, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1187,7 +939,6 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true,
|
||||
.SupportsFlags = true,
|
||||
.SupportsSaturatingRoundingShifts = true,
|
||||
.SupportsVTBL2 = true,
|
||||
|
||||
@@ -26,6 +26,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -43,7 +44,7 @@ public:
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -58,6 +59,8 @@ private:
|
||||
|
||||
const bool HostSupportsSVE128{};
|
||||
const bool HostSupportsSVE256{};
|
||||
const bool HostSupportsRPRES{};
|
||||
const bool HostSupportsAFP{};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
@@ -113,6 +116,25 @@ private:
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Src, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
|
||||
return ARMEmitter::Reg::zr;
|
||||
} else {
|
||||
return GetReg(Src.ID());
|
||||
}
|
||||
}
|
||||
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
ARMEmitter::ShiftType::ROR;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
@@ -208,288 +230,30 @@ private:
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
OpType RT_StoreRegister;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
///< No-op Handler
|
||||
DEF_OP(NoOp);
|
||||
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(AddNZCV);
|
||||
DEF_OP(AdcNZCV);
|
||||
DEF_OP(SbbNZCV);
|
||||
DEF_OP(TestNZ);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(SubNZCV);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
DEF_OP(UDiv);
|
||||
DEF_OP(Rem);
|
||||
DEF_OP(URem);
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(Ornror);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
DEF_OP(Ashr);
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
DEF_OP(LURem);
|
||||
DEF_OP(Zext);
|
||||
DEF_OP(Not);
|
||||
DEF_OP(Popcount);
|
||||
DEF_OP(FindLSB);
|
||||
DEF_OP(FindMSB);
|
||||
DEF_OP(FindTrailingZeroes);
|
||||
DEF_OP(CountLeadingZeroes);
|
||||
DEF_OP(Rev);
|
||||
DEF_OP(Bfi);
|
||||
DEF_OP(Bfxil);
|
||||
DEF_OP(Bfe);
|
||||
DEF_OP(Sbfe);
|
||||
DEF_OP(Select);
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
///< Atomic ops
|
||||
DEF_OP(CASPair);
|
||||
DEF_OP(CAS);
|
||||
DEF_OP(AtomicAdd);
|
||||
DEF_OP(AtomicSub);
|
||||
DEF_OP(AtomicAnd);
|
||||
DEF_OP(AtomicOr);
|
||||
DEF_OP(AtomicXor);
|
||||
DEF_OP(AtomicSwap);
|
||||
DEF_OP(AtomicFetchAdd);
|
||||
DEF_OP(AtomicFetchSub);
|
||||
DEF_OP(AtomicFetchAnd);
|
||||
DEF_OP(AtomicFetchCLR);
|
||||
DEF_OP(AtomicFetchOr);
|
||||
DEF_OP(AtomicFetchXor);
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
DEF_OP(TelemetrySetValue);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(VDupFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
DEF_OP(FillRegister);
|
||||
DEF_OP(LoadFlag);
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(LoadMemTSO);
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(VLoadVectorElement);
|
||||
DEF_OP(VStoreVectorElement);
|
||||
DEF_OP(VBroadcastFromMem);
|
||||
DEF_OP(Push);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(LoadNamedVectorConstant);
|
||||
DEF_OP(LoadNamedVectorIndexedConstant);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
DEF_OP(VSub);
|
||||
DEF_OP(VUQAdd);
|
||||
DEF_OP(VUQSub);
|
||||
DEF_OP(VSQAdd);
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VFAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
DEF_OP(VFMul);
|
||||
DEF_OP(VFDiv);
|
||||
DEF_OP(VFMin);
|
||||
DEF_OP(VFMax);
|
||||
DEF_OP(VFRecp);
|
||||
DEF_OP(VFSqrt);
|
||||
DEF_OP(VFRSqrt);
|
||||
DEF_OP(VNeg);
|
||||
DEF_OP(VFNeg);
|
||||
DEF_OP(VNot);
|
||||
DEF_OP(VUMin);
|
||||
DEF_OP(VSMin);
|
||||
DEF_OP(VUMax);
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VTrn);
|
||||
DEF_OP(VTrn2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
DEF_OP(VCMPGT);
|
||||
DEF_OP(VCMPGTZ);
|
||||
DEF_OP(VCMPLTZ);
|
||||
DEF_OP(VFCMPEQ);
|
||||
DEF_OP(VFCMPNEQ);
|
||||
DEF_OP(VFCMPLT);
|
||||
DEF_OP(VFCMPGT);
|
||||
DEF_OP(VFCMPLE);
|
||||
DEF_OP(VFCMPORD);
|
||||
DEF_OP(VFCMPUNO);
|
||||
DEF_OP(VUShl);
|
||||
DEF_OP(VUShr);
|
||||
DEF_OP(VSShr);
|
||||
DEF_OP(VUShlS);
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VUShrSWide);
|
||||
DEF_OP(VSShrSWide);
|
||||
DEF_OP(VUShlSWide);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
DEF_OP(VUXTL2);
|
||||
DEF_OP(VSQXTN);
|
||||
DEF_OP(VSQXTN2);
|
||||
DEF_OP(VSQXTNPair);
|
||||
DEF_OP(VSQXTUN);
|
||||
DEF_OP(VSQXTUN2);
|
||||
DEF_OP(VSQXTUNPair);
|
||||
DEF_OP(VSRSHR);
|
||||
DEF_OP(VSQSHL);
|
||||
DEF_OP(VMul);
|
||||
DEF_OP(VUMull);
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUMulH);
|
||||
DEF_OP(VSMulH);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VTBL2);
|
||||
DEF_OP(VRev32);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VFCADD);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
DEF_OP(AESEnc);
|
||||
DEF_OP(AESEncLast);
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#define IROP_DISPATCH_DEFS
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
@@ -131,171 +132,11 @@ DEF_OP(LoadRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
const auto regId =
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ldrb(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldrh(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).W(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrb(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrh(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 8: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
strb(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
strh(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.W(), STATE, Op->Offset);
|
||||
break;
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs: {}", regOffs);
|
||||
strh(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
@@ -334,7 +175,7 @@ DEF_OP(LoadRegisterSRA) {
|
||||
if (HostSupportsSVE256) {
|
||||
const auto regOffs = Op->Offset & 31;
|
||||
|
||||
ARMEmitter::ForwardLabel DataLocation;
|
||||
ARMEmitter::SingleUseForwardLabel DataLocation;
|
||||
const auto LoadPredicate = [this, &DataLocation] {
|
||||
const auto Predicate = ARMEmitter::PReg::p0;
|
||||
adr(TMP1, &DataLocation);
|
||||
@@ -343,7 +184,7 @@ DEF_OP(LoadRegisterSRA) {
|
||||
};
|
||||
|
||||
const auto EmitData = [this, &DataLocation](uint32_t Value) {
|
||||
ARMEmitter::ForwardLabel PastConstant;
|
||||
ARMEmitter::SingleUseForwardLabel PastConstant;
|
||||
b(&PastConstant);
|
||||
Bind(&DataLocation);
|
||||
dc32(Value);
|
||||
@@ -468,15 +309,19 @@ DEF_OP(LoadRegisterSRA) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegisterSRA) {
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
const auto regId =
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
@@ -519,7 +364,7 @@ DEF_OP(StoreRegisterSRA) {
|
||||
const auto regOffs = Op->Offset & 31;
|
||||
|
||||
// Compartmentalized setting up of the predicate for the cases that need it.
|
||||
ARMEmitter::ForwardLabel DataLocation;
|
||||
ARMEmitter::SingleUseForwardLabel DataLocation;
|
||||
const auto LoadPredicate = [this, &DataLocation] {
|
||||
const auto Predicate = ARMEmitter::PReg::p0;
|
||||
adr(TMP1, &DataLocation);
|
||||
@@ -532,7 +377,7 @@ DEF_OP(StoreRegisterSRA) {
|
||||
// It's helpful to treat LoadPredicate and EmitData as a prologue and epilogue
|
||||
// respectfully.
|
||||
const auto EmitData = [this, &DataLocation](uint32_t Data) {
|
||||
ARMEmitter::ForwardLabel PastConstant;
|
||||
ARMEmitter::SingleUseForwardLabel PastConstant;
|
||||
b(&PastConstant);
|
||||
Bind(&DataLocation);
|
||||
dc32(Data);
|
||||
@@ -1031,23 +876,37 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadNZCV) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(StoreNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_StoreNZCV>();
|
||||
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
|
||||
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
|
||||
"PF/AF must be accessed as registers");
|
||||
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
|
||||
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
|
||||
"PF/AF must be accessed as registers");
|
||||
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
|
||||
@@ -1520,13 +1379,13 @@ DEF_OP(VBroadcastFromMem) {
|
||||
ElementSize == 4 || ElementSize == 8 ||
|
||||
ElementSize == 16, "Invalid element size");
|
||||
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use SVE 256-bit broadcast");
|
||||
}
|
||||
if (Is256Bit && !HostSupportsSVE256) {
|
||||
LOGMAN_MSG_A_FMT("{}: 256-bit vectors must support SVE256", __func__);
|
||||
return;
|
||||
}
|
||||
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B.Zeroing()
|
||||
: PRED_TMP_16B.Zeroing();
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
@@ -1840,9 +1699,15 @@ DEF_OP(MemSet) {
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
// If Direction == 0 then:
|
||||
// MemReg is incremented (by size)
|
||||
// else:
|
||||
@@ -1850,8 +1715,8 @@ DEF_OP(MemSet) {
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::ForwardLabel BackwardImpl{};
|
||||
ARMEmitter::ForwardLabel Done{};
|
||||
ARMEmitter::SingleUseForwardLabel BackwardImpl{};
|
||||
ARMEmitter::SingleUseForwardLabel Done{};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
if (Op->Prefix.IsInvalid()) {
|
||||
@@ -1862,8 +1727,10 @@ DEF_OP(MemSet) {
|
||||
add(TMP2, Prefix.X(), MemReg.X());
|
||||
}
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
@@ -1917,13 +1784,12 @@ DEF_OP(MemSet) {
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
auto EmitMemset = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
ARMEmitter::BackwardLabel AgainInternal{};
|
||||
ARMEmitter::ForwardLabel DoneInternal{};
|
||||
ARMEmitter::SingleUseForwardLabel DoneInternal{};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
@@ -1978,15 +1844,26 @@ DEF_OP(MemSet) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
if (DirectionIsInline) {
|
||||
// If the direction constant is set then the direction is negative.
|
||||
EmitMemset(DirectionConstant ? -1 : 1);
|
||||
}
|
||||
else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
EmitMemset(Direction);
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
@@ -2001,7 +1878,12 @@ DEF_OP(MemCpy) {
|
||||
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
auto Dst = GetRegPair(Node);
|
||||
// If Direction == 0 then:
|
||||
@@ -2013,8 +1895,8 @@ DEF_OP(MemCpy) {
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::ForwardLabel BackwardImpl{};
|
||||
ARMEmitter::ForwardLabel Done{};
|
||||
ARMEmitter::SingleUseForwardLabel BackwardImpl{};
|
||||
ARMEmitter::SingleUseForwardLabel Done{};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
if (Op->PrefixDest.IsInvalid()) {
|
||||
@@ -2038,8 +1920,10 @@ DEF_OP(MemCpy) {
|
||||
// TMP3 = Src
|
||||
// TMP4 = load+store temp value
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
@@ -2161,13 +2045,12 @@ DEF_OP(MemCpy) {
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
auto EmitMemcpy = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
ARMEmitter::BackwardLabel AgainInternal{};
|
||||
ARMEmitter::ForwardLabel DoneInternal{};
|
||||
ARMEmitter::SingleUseForwardLabel DoneInternal{};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
@@ -2235,15 +2118,24 @@ DEF_OP(MemCpy) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
if (DirectionIsInline) {
|
||||
// If the direction constant is set then the direction is negative.
|
||||
EmitMemcpy(DirectionConstant ? -1 : 1);
|
||||
}
|
||||
else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
|
||||
@@ -143,7 +143,17 @@ DEF_OP(Print) {
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -15,10 +15,6 @@ struct InternalThreadState;
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures();
|
||||
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
@@ -100,15 +100,15 @@ public:
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
void Erase(uint64_t Address) {
|
||||
void Erase(FEXCore::Core::CpuStateFrame *Frame, uint64_t Address) {
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, 0});
|
||||
auto upper = BlockLinks->upper_bound({Address, UINTPTR_MAX});
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData *>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second();
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
@@ -141,8 +141,7 @@ public:
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
}
|
||||
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData * HostLink, const FEXCore::Context::BlockDelinkerFunc &delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
@@ -224,7 +223,7 @@ private:
|
||||
|
||||
struct BlockLinkTag {
|
||||
uint64_t GuestDestination;
|
||||
uintptr_t HostLink;
|
||||
FEXCore::Context::ExitFunctionLinkData *HostLink;
|
||||
|
||||
bool operator <(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination)
|
||||
@@ -243,7 +242,7 @@ private:
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType *BlockLinks;
|
||||
|
||||
|
||||
@@ -32,9 +32,6 @@ namespace FEXCore::CodeSerialize {
|
||||
// ABI local flag unsafe optimization
|
||||
unsigned ABILocalFlags : 1;
|
||||
|
||||
// Static register allocation enabled
|
||||
unsigned SRA : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
unsigned ParanoidTSO : 1;
|
||||
|
||||
@@ -49,7 +46,7 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
unsigned _Pad : 19;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
@@ -59,7 +56,6 @@ namespace FEXCore::CodeSerialize {
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
SRA == other.SRA &&
|
||||
ParanoidTSO == other.ParanoidTSO &&
|
||||
Is64BitMode == other.Is64BitMode &&
|
||||
SMCChecks == other.SMCChecks &&
|
||||
@@ -75,7 +71,6 @@ namespace FEXCore::CodeSerialize {
|
||||
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.SRA;
|
||||
Hash <<= 1; Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1; Hash |= other.Is64BitMode;
|
||||
Hash <<= 2; Hash |= other.SMCChecks;
|
||||
|
||||
@@ -18,7 +18,6 @@ namespace FEXCore::CodeSerialize {
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -8,7 +8,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
@@ -23,19 +22,34 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Tmp = _Ror(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
|
||||
auto Top = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Src, 3), Tmp);
|
||||
auto Result = _VInsGPR(16, 4, 3, Src, Top);
|
||||
OrderedNode *RotatedNode{};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
// Move the element to zero, rotate, and then move back (Using duplicates).
|
||||
// Saves one instruction versus that path that doesn't support SHA extension.
|
||||
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Sha1HRotated = _VSha1H(Duplicated);
|
||||
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
|
||||
}
|
||||
else {
|
||||
// SHA1 extension missing, manually rotate.
|
||||
// Emulate rotate.
|
||||
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
|
||||
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
|
||||
}
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
|
||||
@@ -46,26 +60,34 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ROR by 31 is equivalent to a ROL by 1
|
||||
auto ThirtyOne = _Constant(32, 31);
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
|
||||
auto W13 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W16 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), W13), ThirtyOne);
|
||||
auto W17 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), W14), ThirtyOne);
|
||||
auto W18 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), W15), ThirtyOne);
|
||||
auto W19 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), W16), ThirtyOne);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W16);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W17);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W18);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W19);
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -81,7 +103,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._And(OpSize::i32Bit, B, D)), Self._And(OpSize::i32Bit, C, D));
|
||||
return Self.BitwiseAtLeastTwo(B, C, D);
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
@@ -102,13 +124,10 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W1 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W2 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto W3 = _VExtractToGPR(16, 4, Src, 0);
|
||||
|
||||
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
|
||||
|
||||
@@ -127,8 +146,12 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C,
|
||||
OrderedNode *D, OrderedNode *E, OrderedNode *W) -> RoundResult {
|
||||
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W), E), K);
|
||||
OrderedNode *D, OrderedNode *E, OrderedNode *Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(16, 4, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto DNext = C;
|
||||
@@ -138,9 +161,9 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
};
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, W1);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, W2);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, W3);
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
|
||||
@@ -151,30 +174,37 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Result{};
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
}
|
||||
else {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
Result = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
@@ -182,8 +212,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))), _Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
@@ -200,74 +230,83 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode *A, OrderedNode *B, OrderedNode *C) {
|
||||
// Returns whether at least 2/3 of A/B/C is true.
|
||||
// Expressed as (A & (B | C)) | (B & C)
|
||||
//
|
||||
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
|
||||
auto And = _And(OpSize::i32Bit, B, C);
|
||||
auto Or = _Or(OpSize::i32Bit, B, C);
|
||||
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Major = [this](OrderedNode *A, OrderedNode *B, OrderedNode *C) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, A, B), _And(OpSize::i32Bit, A, C)), _And(OpSize::i32Bit, B, C));
|
||||
};
|
||||
const auto Sigma0 = [this](OrderedNode *A) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), _Ror(OpSize::i32Bit, A, _Constant(32, 13))), _Ror(OpSize::i32Bit, A, _Constant(32, 22)));
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](OrderedNode *E) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), _Ror(OpSize::i32Bit, E, _Constant(32, 11))), _Ror(OpSize::i32Bit, E, _Constant(32, 25)));
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
OrderedNode *Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
|
||||
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
|
||||
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
OrderedNode * Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
|
||||
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*,
|
||||
OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
|
||||
const auto Round = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C, OrderedNode *D,
|
||||
OrderedNode *E, OrderedNode *F, OrderedNode *G, OrderedNode *H,
|
||||
OrderedNode* WK) -> RoundResult {
|
||||
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), Major(A, B, C)), Sigma0(A));
|
||||
auto BNext = A;
|
||||
auto CNext = B;
|
||||
auto DNext = C;
|
||||
auto ENext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), D);
|
||||
auto FNext = E;
|
||||
auto GNext = F;
|
||||
auto HNext = G;
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
|
||||
return {ANext, BNext, CNext, DNext, ENext, FNext, GNext, HNext};
|
||||
};
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
|
||||
auto [A1, B1, C1, D1, E1, F1, G1, H1] = Round(A0, B0, C0, D0, E0, F0, G0, H0, WK0);
|
||||
auto Final = Round(A1, B1, C1, D1, E1, F1, G1, H1, WK1);
|
||||
|
||||
auto Res3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Res2 = _VInsGPR(16, 4, 2, Res3, std::get<1>(Final));
|
||||
auto Res1 = _VInsGPR(16, 4, 1, Res2, std::get<4>(Final));
|
||||
auto Res0 = _VInsGPR(16, 4, 0, Res1, std::get<5>(Final));
|
||||
auto Res3 = _VInsGPR(16, 4, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(16, 4, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(16, 4, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(16, 4, 0, Res1, E1);
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -280,8 +319,8 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -289,8 +328,8 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -303,8 +342,8 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -312,8 +351,8 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -326,8 +365,8 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -335,8 +374,8 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -349,8 +388,8 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -358,7 +397,7 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
@@ -375,8 +414,8 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector);
|
||||
@@ -388,8 +427,8 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
OrderedNode *Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -14,7 +14,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
@@ -139,7 +138,7 @@ void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
}
|
||||
else {
|
||||
// Implicit arg
|
||||
@@ -178,7 +177,7 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1);
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
_StoreContextIndexed(converted, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -238,7 +237,7 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
|
||||
auto zero = _Constant(0);
|
||||
|
||||
@@ -247,10 +246,13 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
|
||||
// Extract sign and make interger absolute
|
||||
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
|
||||
// We're about to clobber flags to grab the sign, so save NZCV.
|
||||
SaveNZCV();
|
||||
|
||||
auto absolute = _Abs(OpSize::i64Bit, data);
|
||||
// Extract sign and make interger absolute
|
||||
_SubNZCV(OpSize::i64Bit, data, zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType{COND_SLT}, _Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, data, CondClassType{COND_MI});
|
||||
|
||||
// left justify the absolute interger
|
||||
auto shift = _Sub(OpSize::i64Bit, _Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
@@ -334,11 +336,11 @@ void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -395,11 +397,11 @@ void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -458,11 +460,11 @@ void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -543,11 +545,11 @@ void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -724,11 +726,11 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -763,13 +765,13 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -856,9 +858,7 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80Round(a);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F80Round(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
@@ -889,9 +889,7 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80Add(a, st1);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F80Add(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
@@ -1014,7 +1012,7 @@ void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
@@ -1052,7 +1050,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
@@ -1099,7 +1097,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
@@ -1110,7 +1108,7 @@ void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDSW(OpcodeArgs) {
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
}
|
||||
|
||||
@@ -1139,7 +1137,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
@@ -1213,7 +1211,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
|
||||
@@ -13,7 +13,6 @@ $end_info$
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
@@ -66,7 +65,7 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
@@ -89,7 +88,7 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
@@ -112,7 +111,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
if constexpr (width == 32) {
|
||||
converted = _Float_FToF(8, 4, data);
|
||||
@@ -153,7 +152,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1);
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
converted = _F80CVT(8, converted);
|
||||
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -210,7 +209,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
if(read_width == 2) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
@@ -292,16 +291,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -353,16 +352,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -417,16 +416,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -503,16 +502,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -601,21 +600,13 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
auto low = _Constant(0);
|
||||
OrderedNode *data = _VCastFromGPR(8, 8, low);
|
||||
|
||||
OrderedNode *Res = _FCmp(8, a, data,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
// Now we do our comparison.
|
||||
_FCmp(8, a, data);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
|
||||
//TODO: This should obey rounding mode
|
||||
@@ -661,16 +652,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -681,36 +672,22 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode *Res = _FCmp(8, a, b,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToSSE();
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -767,9 +744,7 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64SIN(a);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F64SIN(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
@@ -799,9 +774,7 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64ATAN(a, st1);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F64ATAN(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM ||
|
||||
IROp == IR::OP_F64FPREM1) {
|
||||
@@ -921,7 +894,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
@@ -999,7 +972,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
|
||||
@@ -1,41 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <unistd.h>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
|
||||
// Let the host take first stab at handling the signal
|
||||
auto Thread = GetTLSThread();
|
||||
HostSignalHandler &Handler = HostHandlers[Signal];
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
for (auto &Handler : Handler.Handlers) {
|
||||
if (Handler(Thread, Signal, Info, UContext)) {
|
||||
// If the host handler handled the fault then we can continue now
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Handler.FrontendHandler &&
|
||||
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Now let the frontend handle the signal
|
||||
// It's clearly a guest signal and this ends up being an OS specific issue
|
||||
HandleGuestSignal(Thread, Signal, Info, UContext);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,84 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#ifndef NDEBUG
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::X86Tables::X86InstDebugInfo {
|
||||
void InstallDebugInfo() {
|
||||
const std::tuple<uint8_t, uint8_t, Flags> BaseOpTable[] = {
|
||||
{0x50, 8, {FLAGS_MEM_ACCESS}},
|
||||
{0x58, 8, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0x68, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0x6A, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xAA, 4, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xC8, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xCC, 2, {FLAGS_DEBUG}},
|
||||
|
||||
{0xD7, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xF1, 1, {FLAGS_DEBUG}},
|
||||
{0xF4, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::tuple<uint8_t, uint8_t, Flags> TwoByteOpTable[] = {
|
||||
{0x0B, 1, {FLAGS_DEBUG}},
|
||||
{0x19, 7, {FLAGS_DEBUG}},
|
||||
{0x28, 2, {FLAGS_MEM_ALIGN_16}},
|
||||
|
||||
{0x31, 1, {FLAGS_DEBUG}},
|
||||
|
||||
{0xA2, 1, {FLAGS_DEBUG}},
|
||||
{0xA3, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xAB, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xB3, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xBB, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xFF, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::tuple<uint8_t, uint8_t, Flags> PrimaryGroupOpTable[] = {
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 2, {FLAGS_DIVIDE}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 2, {FLAGS_DIVIDE}},
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
const std::tuple<uint16_t, uint8_t, Flags> SecondaryExtensionOpTable[] = {
|
||||
#define PF_NONE 0
|
||||
#define PF_F3 1
|
||||
#define PF_66 2
|
||||
#define PF_F2 3
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, {FLAGS_DEBUG}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, {FLAGS_DEBUG}},
|
||||
#undef PF_F3
|
||||
#undef PF_66
|
||||
#undef PF_F2
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
auto GenerateDebugTable = [](auto& FinalTable, auto& LocalTable) {
|
||||
for (auto Op : LocalTable) {
|
||||
auto OpNum = std::get<0>(Op);
|
||||
auto DebugInfo = std::get<2>(Op);
|
||||
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
|
||||
memcpy(&FinalTable[OpNum+i].DebugInfo, &DebugInfo, sizeof(X86InstDebugInfo::Flags));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
GenerateDebugTable(BaseOps, BaseOpTable);
|
||||
GenerateDebugTable(SecondBaseOps, TwoByteOpTable);
|
||||
GenerateDebugTable(PrimaryInstGroupOps, PrimaryGroupOpTable);
|
||||
|
||||
GenerateDebugTable(SecondInstGroupOps, SecondaryExtensionOpTable);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
@@ -12,60 +12,16 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps{};
|
||||
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps{};
|
||||
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps{};
|
||||
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps{};
|
||||
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps{};
|
||||
|
||||
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps{};
|
||||
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps{};
|
||||
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps{};
|
||||
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops{};
|
||||
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps{};
|
||||
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps{};
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps{};
|
||||
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps{};
|
||||
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps{};
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps{};
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps{};
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps{};
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode);
|
||||
void InitializePrimaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryGroupTables();
|
||||
void InitializeSecondaryModRMTables();
|
||||
void InitializeX87Tables();
|
||||
void InitializeDDDTables();
|
||||
void InitializeH0F38Tables();
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode);
|
||||
void InitializeVEXTables();
|
||||
void InitializeXOPTables();
|
||||
void InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
uint64_t Total{};
|
||||
uint64_t NumInsts{};
|
||||
#endif
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
InitializePrimaryGroupTables(Mode);
|
||||
InitializeSecondaryGroupTables();
|
||||
InitializeSecondaryModRMTables();
|
||||
InitializeX87Tables();
|
||||
InitializeDDDTables();
|
||||
InitializeH0F38Tables();
|
||||
InitializeH0F3ATables(Mode);
|
||||
InitializeVEXTables();
|
||||
InitializeXOPTables();
|
||||
InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
X86InstDebugInfo::InstallDebugInfo();
|
||||
#endif
|
||||
}
|
||||
|
||||
}
|
||||
@@ -14,8 +14,10 @@ $end_info$
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
static constexpr U8U8InfoStruct BaseOpTable[] = {
|
||||
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> Table{};
|
||||
|
||||
constexpr U8U8InfoStruct BaseOpTable[] = {
|
||||
// Prefixes
|
||||
// Operand size overide
|
||||
{0x66, 1, X86InstInfo{"", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -100,10 +102,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
|
||||
// This should just throw a GP
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
@@ -147,19 +149,19 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1, nullptr}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4, nullptr}},
|
||||
@@ -169,7 +171,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3, nullptr}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
@@ -192,8 +194,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -233,6 +235,12 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC4, 2, X86InstInfo{"", TYPE_VEX_TABLE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), BaseOpTable, std::size(BaseOpTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
static constexpr U8U8InfoStruct BaseOpTable_64[] = {
|
||||
{0x06, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x0E, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -291,8 +299,6 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable, std::size(BaseOpTable));
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_64, std::size(BaseOpTable_64));
|
||||
}
|
||||
|
||||
@@ -12,8 +12,9 @@ $end_info$
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeDDDTables() {
|
||||
static constexpr U8U8InfoStruct DDDNowOpTable[] = {
|
||||
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> Table{};
|
||||
constexpr U8U8InfoStruct DDDNowOpTable[] = {
|
||||
{0x0C, 1, X86InstInfo{"PI2FW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x0D, 1, X86InstInfo{"PI2FD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x1C, 1, X86InstInfo{"PF2IW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
@@ -52,6 +53,8 @@ void InitializeDDDTables() {
|
||||
{0xBF, 1, X86InstInfo{"PAVGUSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&DDDNowOps.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
}
|
||||
GenerateTable(&Table.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -11,9 +11,9 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeEVEXTables() {
|
||||
static constexpr U16U8InfoStruct EVEXTable[] = {
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct EVEXTable[] = {
|
||||
{0x10, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x11, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"VBROADCASTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -29,6 +29,9 @@ void InitializeEVEXTables() {
|
||||
{0xE7, 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&EVEXTableOps.at(0), EVEXTable, std::size(EVEXTable));
|
||||
}
|
||||
GenerateTable(&Table.at(0), EVEXTable, std::size(EVEXTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -12,15 +12,16 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> Table{};
|
||||
|
||||
void InitializeH0F38Tables() {
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
static constexpr U16U8InfoStruct H0F38Table[] = {
|
||||
constexpr U16U8InfoStruct H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x01), 1, X86InstInfo{"PHADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
@@ -117,6 +118,8 @@ void InitializeH0F38Tables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&H0F38TableOps.at(0), H0F38Table, std::size(H0F38Table));
|
||||
}
|
||||
GenerateTable(&Table.at(0), H0F38Table, std::size(H0F38Table));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -14,13 +14,13 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
constexpr uint16_t PF_3A_NONE = 0;
|
||||
constexpr uint16_t PF_3A_66 = 1;
|
||||
constexpr uint16_t PF_3A_NONE = 0;
|
||||
constexpr uint16_t PF_3A_66 = 1;
|
||||
|
||||
static constexpr U16U8InfoStruct H0F3ATable[] = {
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
@@ -54,6 +54,11 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
static constexpr U16U8InfoStruct H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
@@ -62,8 +67,6 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable_64, std::size(H0F3ATable_64));
|
||||
}
|
||||
|
||||
@@ -13,10 +13,10 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
|
||||
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> Table{};
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
const U16U8InfoStruct PrimaryGroupOpTable[] = {
|
||||
constexpr U16U8InfoStruct PrimaryGroupOpTable[] = {
|
||||
// GROUP_1 | 0x80 | reg
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 0), 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, nullptr}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 1), 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, nullptr}},
|
||||
@@ -141,9 +141,13 @@ void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 7), 1, X86InstInfo{"XBEGIN", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_SETS_RIP | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
|
||||
const U16U8InfoStruct PrimaryGroupOpTable_64[] = {
|
||||
// Invalid in 64bit mode
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 0), 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -163,7 +167,6 @@ void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
|
||||
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable_64, std::size(PrimaryGroupOpTable_64));
|
||||
}
|
||||
|
||||
@@ -12,15 +12,15 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeSecondaryGroupTables() {
|
||||
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> Table{};
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
constexpr uint16_t PF_NONE = 0;
|
||||
constexpr uint16_t PF_F3 = 1;
|
||||
constexpr uint16_t PF_66 = 2;
|
||||
constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
static constexpr U16U8InfoStruct SecondaryExtensionOpTable[] = {
|
||||
constexpr U16U8InfoStruct SecondaryExtensionOpTable[] = {
|
||||
// GROUP 1
|
||||
// GROUP 2
|
||||
// GROUP 3
|
||||
@@ -183,41 +183,41 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 10
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
// GROUP 12
|
||||
{OPD(TYPE_GROUP_12, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -487,7 +487,8 @@ void InitializeSecondaryGroupTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&SecondInstGroupOps.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
|
||||
}
|
||||
GenerateTable(&Table.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -11,9 +11,9 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeSecondaryModRMTables() {
|
||||
static constexpr U8U8InfoStruct SecondaryModRMExtensionOpTable[] = {
|
||||
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> Table{};
|
||||
constexpr U8U8InfoStruct SecondaryModRMExtensionOpTable[] = {
|
||||
// REG /1
|
||||
{((0 << 3) | 0), 1, X86InstInfo{"MONITOR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 1), 1, X86InstInfo{"MWAIT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -55,6 +55,8 @@ void InitializeSecondaryModRMTables() {
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&SecondModRMTableOps.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
|
||||
}
|
||||
GenerateTable(&Table.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -13,9 +13,10 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
auto BaseOpsLambda = []() consteval {
|
||||
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> Table{};
|
||||
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
static constexpr U8U8InfoStruct TwoByteOpTable[] = {
|
||||
constexpr U8U8InfoStruct TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x00, 1, X86InstInfo{"", TYPE_GROUP_6, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x01, 1, X86InstInfo{"", TYPE_GROUP_7, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -29,7 +30,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0C, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0D, 1, X86InstInfo{"", TYPE_GROUP_P, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0E, 1, X86InstInfo{"FEMMS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -44,7 +45,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x16, 1, X86InstInfo{"MOVLHPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x17, 1, X86InstInfo{"MOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"", TYPE_GROUP_16, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_DEBUG | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0x20, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x22, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -59,7 +60,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x2F, 1, X86InstInfo{"COMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"WRMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_DEBUG | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x32, 1, X86InstInfo{"RDMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x33, 1, X86InstInfo{"RDPMC", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -166,11 +167,13 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SETLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"SETNLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_DEBUG | FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA0, 2, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA6, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA8, 2, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"RSM", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
|
||||
@@ -254,7 +257,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xFC, 1, X86InstInfo{"PADDB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFD, 1, X86InstInfo{"PADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
@@ -265,23 +268,16 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0, nullptr}},
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct TwoByteOpTable_32[] = {
|
||||
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
};
|
||||
return Table;
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct TwoByteOpTable_64[] = {
|
||||
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = BaseOpsLambda();
|
||||
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> Table{};
|
||||
|
||||
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct RepModOpTable[] = {
|
||||
constexpr U8U8InfoStruct RepModOpTable[] = {
|
||||
{0x0, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"MOVSS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -361,7 +357,15 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xFF, 1, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct RepNEModOpTable[] = {
|
||||
GenerateTableWithCopy(&Table.at(0), RepModOpTable, std::size(RepModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> Table{};
|
||||
|
||||
constexpr U8U8InfoStruct RepNEModOpTable[] = {
|
||||
{0x0, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"MOVSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -434,7 +438,15 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xF8, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct OpSizeModOpTable[] = {
|
||||
GenerateTableWithCopy(&Table.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> Table{};
|
||||
|
||||
constexpr U8U8InfoStruct OpSizeModOpTable[] = {
|
||||
{0x0, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"MOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -581,19 +593,40 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xFF, 1, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
GenerateTableWithCopy(&Table.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &BaseOpsLambda().at(0));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
static constexpr U8U8InfoStruct TwoByteOpTable_32[] = {
|
||||
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct TwoByteOpTable_64[] = {
|
||||
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
};
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&SecondBaseOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
}
|
||||
else {
|
||||
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
}
|
||||
|
||||
GenerateTableWithCopy(&RepModOps.at(0), RepModOpTable, std::size(RepModOpTable), &SecondBaseOps.at(0));
|
||||
GenerateTableWithCopy(&RepNEModOps.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &SecondBaseOps.at(0));
|
||||
GenerateTableWithCopy(&OpSizeModOps.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &SecondBaseOps.at(0));
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
@@ -11,10 +11,10 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeVEXTables() {
|
||||
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> Table{};
|
||||
#define OPD(map_select, pp, opcode) (((map_select - 1) << 10) | (pp << 8) | (opcode))
|
||||
static constexpr U16U8InfoStruct VEXTable[] = {
|
||||
constexpr U16U8InfoStruct VEXTable[] = {
|
||||
// Map 0 (Reserved)
|
||||
// VEX Map 1
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -488,8 +488,15 @@ void InitializeVEXTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), VEXTable, std::size(VEXTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> Table{};
|
||||
|
||||
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
static constexpr U8U8InfoStruct VEXGroupTable[] = {
|
||||
constexpr U8U8InfoStruct VEXGroupTable[] = {
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b010), 1, X86InstInfo{"VPSRLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b100), 1, X86InstInfo{"VPSRAW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(TYPE_VEX_GROUP_12, 1, 0b110), 1, X86InstInfo{"VPSLLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
@@ -512,7 +519,8 @@ void InitializeVEXTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&VEXTableOps.at(0), VEXTable, std::size(VEXTable));
|
||||
GenerateTable(&VEXTableGroupOps.at(0), VEXGroupTable, std::size(VEXGroupTable));
|
||||
}
|
||||
GenerateTable(&Table.at(0), VEXGroupTable, std::size(VEXGroupTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
}
|
||||
@@ -279,9 +279,15 @@ namespace InstFlags {
|
||||
using InstFlagType = uint64_t;
|
||||
|
||||
constexpr InstFlagType FLAGS_NONE = 0;
|
||||
constexpr InstFlagType FLAGS_DEBUG = (1ULL << 1);
|
||||
// The secondary Opcode Map uses prefix bytes to overlay more instruction
|
||||
// But some instructions need to ignore this overlay and consume these prefixes.
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 0);
|
||||
// Some instructions partially ignore overlay
|
||||
// Ignore OpSize (0x66) in this case
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 1);
|
||||
constexpr InstFlagType FLAGS_DEBUG_MEM_ACCESS = (1ULL << 2);
|
||||
constexpr InstFlagType FLAGS_SUPPORTS_REP = (1ULL << 3);
|
||||
// Only SEXT if the instruction is operating in 64bit operand size
|
||||
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 3);
|
||||
constexpr InstFlagType FLAGS_BLOCK_END = (1ULL << 4);
|
||||
constexpr InstFlagType FLAGS_SETS_RIP = (1ULL << 5);
|
||||
|
||||
@@ -331,27 +337,17 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
|
||||
constexpr InstFlagType FLAGS_SF_MOD_MEM_ONLY = (1ULL << 18);
|
||||
constexpr InstFlagType FLAGS_SF_MOD_REG_ONLY = (1ULL << 19);
|
||||
|
||||
// The secondary Opcode Map uses prefix bytes to overlay more instruction
|
||||
// But some instructions need to ignore this overlay and consume these prefixes.
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 20);
|
||||
// Some instructions partially ignore overlay
|
||||
// Ignore OpSize (0x66) in this case
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 21);
|
||||
|
||||
// x87
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 22);
|
||||
|
||||
// Only SEXT if the instruction is operating in 64bit operand size
|
||||
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 23);
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
|
||||
|
||||
// Whether or not the instruction has a VEX prefix for the first source operand
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
|
||||
// Whether or not the instruction has a VEX prefix for the second source operand
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -419,35 +415,12 @@ constexpr uint8_t OpToIndex(uint8_t Op) {
|
||||
using DecodedOp = DecodedInst const*;
|
||||
using OpDispatchPtr = void (IR::OpDispatchBuilder::*)(DecodedOp);
|
||||
|
||||
#ifndef NDEBUG
|
||||
namespace X86InstDebugInfo {
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_4 = (1 << 0);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_8 = (1 << 1);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_16 = (1 << 2);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_SIZE = (1 << 3); // If instruction size changes depending on prefixes
|
||||
constexpr uint64_t FLAGS_MEM_ACCESS = (1 << 4);
|
||||
constexpr uint64_t FLAGS_DEBUG = (1 << 5);
|
||||
constexpr uint64_t FLAGS_DIVIDE = (1 << 6);
|
||||
|
||||
|
||||
struct Flags {
|
||||
uint64_t DebugFlags;
|
||||
};
|
||||
void InstallDebugInfo();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
struct X86InstInfo {
|
||||
char const *Name;
|
||||
InstType Type;
|
||||
InstFlags::InstFlagType Flags; ///< Must be larger than InstFlags enum
|
||||
uint8_t MoreBytes;
|
||||
OpDispatchPtr OpcodeDispatcher;
|
||||
#ifndef NDEBUG
|
||||
X86InstDebugInfo::Flags DebugInfo;
|
||||
uint32_t NumUnitTestsGenerated;
|
||||
#endif
|
||||
|
||||
bool operator==(const X86InstInfo &b) const {
|
||||
if (strcmp(Name, b.Name) != 0 ||
|
||||
@@ -524,12 +497,6 @@ extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
// EVEX
|
||||
extern std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
|
||||
|
||||
|
||||
#ifndef NDEBUG
|
||||
extern uint64_t Total;
|
||||
extern uint64_t NumInsts;
|
||||
#endif
|
||||
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
@@ -540,54 +507,65 @@ using U8U8InfoStruct = X86TablesInfoStruct<uint8_t>;
|
||||
using U16U8InfoStruct = X86TablesInfoStruct<uint16_t>;
|
||||
|
||||
template<typename OpcodeType>
|
||||
static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
constexpr static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template<typename OpcodeType>
|
||||
static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize, X86InstInfo *OtherLocal) {
|
||||
constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize, X86InstInfo *OtherLocal) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if (Info.Type == TYPE_COPY_OTHER) {
|
||||
FinalTable[OpNum + i] = OtherLocal[OpNum + i];
|
||||
}
|
||||
else {
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template<typename OpcodeType>
|
||||
static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
static inline void LateInitCopyTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *OtherLocal, size_t OtherTableSize) {
|
||||
for (size_t j = 0; j < OtherTableSize; ++j) {
|
||||
X86TablesInfoStruct<OpcodeType> const &OtherOp = OtherLocal[j];
|
||||
auto OtherOpNum = OtherOp.first;
|
||||
X86InstInfo const &OtherInfo = OtherOp.Info;
|
||||
for (uint32_t i = 0; i < OtherOp.second; ++i) {
|
||||
X86InstInfo &FinalOp = FinalTable[OtherOpNum + i];
|
||||
if (FinalOp.Type == TYPE_COPY_OTHER) {
|
||||
FinalOp = OtherInfo;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename OpcodeType>
|
||||
constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
}
|
||||
if ((OpNum & 0b11'000'000) == 0b11'000'000) {
|
||||
// If the mod field is 0b11 then it is a regular op
|
||||
FinalTable[OpNum + i] = Info;
|
||||
@@ -595,18 +573,15 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct
|
||||
else {
|
||||
// If the mod field is !0b11 then this instruction is duplicated through the whole mod [0b00, 0b10] range
|
||||
// and the modrm.rm space because that is used part of the instruction encoding
|
||||
LOGMAN_THROW_AA_FMT((OpNum & 0b11'000'000) == 0, "Only support mod field of zero in this path");
|
||||
if ((OpNum & 0b11'000'000) != 0) {
|
||||
ERROR_AND_DIE_FMT("Only support mod field of zero in this path");
|
||||
}
|
||||
for (uint16_t mod = 0b00'000'000; mod < 0b11'000'000; mod += 0b01'000'000) {
|
||||
for (uint16_t rm = 0b000; rm < 0b1'000; ++rm) {
|
||||
FinalTable[(OpNum | mod | rm) + i] = Info;
|
||||
}
|
||||
}
|
||||
}
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -11,11 +11,11 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeX87Tables() {
|
||||
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops = []() consteval {
|
||||
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> Table{};
|
||||
#define OPD(op, modrmop) (((op - 0xD8) << 8) | modrmop)
|
||||
#define OPDReg(op, reg) (((op - 0xD8) << 8) | (reg << 3))
|
||||
static constexpr U16U8InfoStruct X87OpTable[] = {
|
||||
constexpr U16U8InfoStruct X87OpTable[] = {
|
||||
// 0xD8
|
||||
{OPDReg(0xD8, 0), 1, X86InstInfo{"FADD", TYPE_X87, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPDReg(0xD8, 1), 1, X86InstInfo{"FMUL", TYPE_X87, FLAGS_MODRM, 0, nullptr}},
|
||||
@@ -263,6 +263,8 @@ void InitializeX87Tables() {
|
||||
#undef OPD
|
||||
#undef OPDReg
|
||||
|
||||
GenerateX87Table(&X87Ops.at(0), X87OpTable, std::size(X87OpTable));
|
||||
}
|
||||
GenerateX87Table(&Table.at(0), X87OpTable, std::size(X87OpTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -12,14 +12,14 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeXOPTables() {
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> Table{};
|
||||
#define OPD(group, pp, opcode) ( (group << 10) | (pp << 8) | (opcode))
|
||||
constexpr uint16_t XOP_GROUP_8 = 0;
|
||||
constexpr uint16_t XOP_GROUP_9 = 1;
|
||||
constexpr uint16_t XOP_GROUP_A = 2;
|
||||
|
||||
static constexpr U16U8InfoStruct XOPTable[] = {
|
||||
constexpr U16U8InfoStruct XOPTable[] = {
|
||||
// Group 8
|
||||
{OPD(XOP_GROUP_8, 0, 0x85), 1, X86InstInfo{"VPMAXSSWW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x86), 1, X86InstInfo{"VPMACSSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -104,8 +104,15 @@ void InitializeXOPTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), XOPTable, std::size(XOPTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> Table{};
|
||||
#define OPD(subgroup, opcode) (((subgroup - 1) << 3) | (opcode))
|
||||
static constexpr U8U8InfoStruct XOPGroupTable[] = {
|
||||
constexpr U8U8InfoStruct XOPGroupTable[] = {
|
||||
// Group 1
|
||||
{OPD(1, 1), 1, X86InstInfo{"BLCFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 2), 1, X86InstInfo{"BLSFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -129,7 +136,8 @@ void InitializeXOPTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&XOPTableOps.at(0), XOPTable, std::size(XOPTable));
|
||||
GenerateTable(&XOPTableGroupOps.at(0), XOPGroupTable, std::size(XOPGroupTable));
|
||||
}
|
||||
GenerateTable(&Table.at(0), XOPGroupTable, std::size(XOPGroupTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -6,12 +6,13 @@ tags: glue|thunks
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -79,7 +80,7 @@ namespace FEXCore {
|
||||
const char *Name;
|
||||
};
|
||||
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread;
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread = nullptr;
|
||||
|
||||
|
||||
struct ExportEntry { uint8_t *sha256; ThunkedFunction* Fn; };
|
||||
@@ -171,8 +172,21 @@ namespace FEXCore {
|
||||
Set arg0/1 to arg regs, use CTX::HandleCallback to handle the callback
|
||||
*/
|
||||
static void CallCallback(void *callback, void *arg0, void* arg1) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to invoke guest callback asynchronously");
|
||||
}
|
||||
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
} else {
|
||||
if ((reinterpret_cast<uintptr_t>(arg1) >> 32) != 0) {
|
||||
ERROR_AND_DIE_FMT("Tried to call guest function with arguments packed to a 64-bit address");
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = (uintptr_t)arg1;
|
||||
}
|
||||
|
||||
Thread->CTX->HandleCallback(Thread, (uintptr_t)callback);
|
||||
}
|
||||
@@ -206,7 +220,7 @@ namespace FEXCore {
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}",
|
||||
args->original_callee, args->target_addr);
|
||||
|
||||
auto Result = Thread->CTX->AddCustomIREntrypoint(
|
||||
auto Result = CTX->AddCustomIREntrypoint(
|
||||
args->original_callee,
|
||||
[CTX, GuestThunkEntrypoint = args->target_addr](uintptr_t Entrypoint, FEXCore::IR::IREmitter *emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
@@ -220,7 +234,7 @@ namespace FEXCore {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
}
|
||||
else {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, mm[0][0]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
@@ -473,6 +487,26 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void* GetGuestStack() {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
|
||||
}
|
||||
|
||||
return (void*)(uintptr_t)((Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP]));
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void MoveGuestStack(uintptr_t NewAddress) {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
|
||||
}
|
||||
|
||||
if (NewAddress >> 32) {
|
||||
ERROR_AND_DIE_FMT("Tried to set stack pointer for 32-bit guest to a 64-bit address");
|
||||
}
|
||||
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = NewAddress;
|
||||
}
|
||||
|
||||
#else
|
||||
fextl::unique_ptr<ThunkHandler> ThunkHandler::Create() {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
|
||||
@@ -363,18 +363,9 @@ namespace FEXCore::IR {
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), std::move(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->DebugStore.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
else {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class RegisterAllocationData;
|
||||
|
||||
class IRListView;
|
||||
class IREmitter;
|
||||
|
||||
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
|
||||
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, fextl::stringstream &MapsStream);
|
||||
}
|
||||
+524
-121
File diff suppressed because it is too large.
Load diff
@@ -44,6 +44,11 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
|
||||
if (Arg == COND_AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
}
|
||||
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
@@ -238,6 +243,18 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
@@ -245,6 +262,16 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::ShiftType Arg) {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
|
||||
@@ -6,8 +6,9 @@ tags: ir|emitter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
@@ -27,6 +27,8 @@ friend class FEXCore::IR::PassManager;
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
virtual ~IREmitter() = default;
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
@@ -332,6 +334,10 @@ friend class FEXCore::IR::PassManager;
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
virtual void SaveNZCV(IROps Op) {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
OrderedNode *CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
@@ -6,12 +6,11 @@ tags: ir|parser
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/StringUtils.h"
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
@@ -506,8 +505,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
if (Def.HasArgs) {
|
||||
RemainingLine = FEXCore::StringUtils::Trim(RemainingLine.substr(CurrentPos));
|
||||
CurrentPos = 0;
|
||||
RemainingLine =
|
||||
FEXCore::StringUtils::Trim(RemainingLine.substr(CurrentPos));
|
||||
if (RemainingLine.empty()) {
|
||||
// How did we get here?
|
||||
Def.HasArgs = false;
|
||||
|
||||
@@ -66,7 +66,7 @@ void PassManager::Finalize() {
|
||||
}
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants, bool StaticRegisterAllocation) {
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
@@ -82,7 +82,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool Inli
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9));
|
||||
|
||||
////// InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
|
||||
InsertPass(CreateInlineCallOptimization(&ctx->CPUID));
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
@@ -101,8 +101,8 @@ void PassManager::AddDefaultValidationPasses() {
|
||||
#endif
|
||||
}
|
||||
|
||||
void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAVX) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), OptimizeSRA, SupportsAVX), "RA");
|
||||
void PassManager::InsertRegisterAllocationPass(bool SupportsAVX) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), SupportsAVX), "RA");
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
|
||||
@@ -29,8 +29,6 @@ namespace FEXCore::IR {
|
||||
class PassManager;
|
||||
class IREmitter;
|
||||
|
||||
using ShouldExitHandler = std::function<void(void)>;
|
||||
|
||||
class Pass {
|
||||
public:
|
||||
virtual ~Pass() = default;
|
||||
@@ -47,7 +45,7 @@ protected:
|
||||
class PassManager final {
|
||||
friend class InlineCallOptimization;
|
||||
public:
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants, bool StaticRegisterAllocation);
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants);
|
||||
void AddDefaultValidationPasses();
|
||||
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
|
||||
auto PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
|
||||
@@ -58,14 +56,10 @@ public:
|
||||
return PassPtr;
|
||||
}
|
||||
|
||||
void InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAVX);
|
||||
void InsertRegisterAllocationPass(bool SupportsAVX);
|
||||
|
||||
bool Run(IREmitter *IREmit);
|
||||
|
||||
void RegisterExitHandler(ShouldExitHandler Handler) {
|
||||
ExitHandler = std::move(Handler);
|
||||
}
|
||||
|
||||
bool HasPass(fextl::string Name) const {
|
||||
return NameToPassMaping.contains(Name);
|
||||
}
|
||||
@@ -86,7 +80,6 @@ public:
|
||||
void Finalize();
|
||||
|
||||
protected:
|
||||
ShouldExitHandler ExitHandler;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler;
|
||||
|
||||
private:
|
||||
|
||||
@@ -24,7 +24,6 @@ fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
|
||||
bool OptimizeSRA,
|
||||
bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
|
||||
@@ -13,10 +13,10 @@ $end_info$
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
@@ -194,8 +194,6 @@ public:
|
||||
|
||||
private:
|
||||
bool HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
@@ -267,91 +265,6 @@ bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& Current
|
||||
return Changed;
|
||||
}
|
||||
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)) {
|
||||
// could be moved
|
||||
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
|
||||
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
|
||||
auto SelectOp = SelectOpHdr->CW<IR::IROp_Select>();
|
||||
|
||||
// the value isn't used after the select otherwise
|
||||
// make sure the sizes match
|
||||
if (SelectOpHdr->Size == UnaryOpHdr->Size && SelectOpHdr->Op == OP_SELECT && SelectOpNode->NumUses == 1
|
||||
&& IREmit->IsValueConstant(SelectOp->TrueVal)
|
||||
&& IREmit->IsValueConstant(SelectOp->FalseVal)) {
|
||||
|
||||
IREmit->SetWriteCursor(IREmit->UnwrapNode(SelectOpNode->Header.Previous));
|
||||
|
||||
size_t OpSize = FEXCore::IR::GetSize(UnaryOpHdr->Op);
|
||||
|
||||
/// copy for TrueVal ///
|
||||
auto NewUnaryOp1 = IREmit->AllocateRawOp(OpSize);
|
||||
|
||||
// Copy over the op
|
||||
memcpy(NewUnaryOp1.first, UnaryOpHdr, OpSize);
|
||||
|
||||
for (int i = 0; i < IR::GetArgs(NewUnaryOp1.first->Op); i++) {
|
||||
NewUnaryOp1.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
|
||||
}
|
||||
// Set New Op to operate on the constant
|
||||
IREmit->ReplaceNodeArgument(NewUnaryOp1, 0, IREmit->UnwrapNode(SelectOp->TrueVal));
|
||||
// Make select use the operated constant
|
||||
IREmit->ReplaceNodeArgument(SelectOpNode, 2, NewUnaryOp1);
|
||||
|
||||
/// copy for FalseVal ///
|
||||
auto NewUnaryOp2 = IREmit->AllocateRawOp(OpSize);
|
||||
|
||||
// Copy over the op
|
||||
memcpy(NewUnaryOp2.first, UnaryOpHdr, OpSize);
|
||||
|
||||
for (int i = 0; i < IR::GetArgs(NewUnaryOp2.first->Op); i++) {
|
||||
NewUnaryOp2.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
|
||||
}
|
||||
// Set New Op to operate on the constant
|
||||
IREmit->ReplaceNodeArgument(NewUnaryOp2, 0, IREmit->UnwrapNode(SelectOp->FalseVal));
|
||||
// Make select use the operated constant
|
||||
IREmit->ReplaceNodeArgument(SelectOpNode, 3, NewUnaryOp2);
|
||||
|
||||
// Replace uses of the defuct unary op w/ select
|
||||
IREmit->ReplaceAllUsesWithRange(UnaryOpNode, SelectOpNode, IREmit->GetIterator(IREmit->WrapNode(UnaryOpNode)), IREmit->GetIterator(BlockOp->Last));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Make all FCMPs set no flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_FCMP) {
|
||||
auto fcmp = IROp->CW<IR::IROp_FCmp>();
|
||||
fcmp->Flags = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Set needed flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_GETHOSTFLAG) {
|
||||
auto ghf = IROp->CW<IR::IROp_GetHostFlag>();
|
||||
|
||||
auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW<IR::IROp_FCmp>();
|
||||
LOGMAN_THROW_AA_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
|
||||
if(fcmp->Header.Op == OP_FCMP) {
|
||||
fcmp->Flags |= 1 << ghf->Flag;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// LoadMem / StoreMem imm pooling
|
||||
// If imms are close by, use address gen to generate the values instead of using a new imm
|
||||
void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
@@ -419,7 +332,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_AND: {
|
||||
// if AND's arguments are imms, they are masking
|
||||
for (int i = 0; i < IR::GetArgs(IROp->Op); i++) {
|
||||
@@ -494,27 +406,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VFADD:
|
||||
case OP_VFSUB:
|
||||
case OP_VFMUL:
|
||||
case OP_VFDIV:
|
||||
case OP_FCMP: {
|
||||
auto flopSize = IROp->Size;
|
||||
for (int i = 0; i < IR::GetArgs(IROp->Op); i++) {
|
||||
auto argHeader = IREmit->GetOpHeader(IROp->Args[i]);
|
||||
|
||||
if (argHeader->Op == OP_VMOV) {
|
||||
auto source = argHeader->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
if (sourceHeader->Size >= flopSize) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(source));
|
||||
//printf("VMOV bypassed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
@@ -668,13 +559,31 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
bool IsConstant1 = IREmit->IsValueConstant(Op->Header.Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(Op->Header.Args[1], &Constant2);
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsConstant1 && IsConstant2) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
else if (IsConstant2 && !IsImmAddSub(Constant2) && IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// This means we can convert the operation in to a subtract.
|
||||
// Change the IR operation itself.
|
||||
IROp->Op = OP_SUB;
|
||||
// Set the write cursor to just before this operation.
|
||||
auto CodeIter = CurrentIR.at(CodeNode);
|
||||
--CodeIter;
|
||||
IREmit->SetWriteCursor(std::get<0>(*CodeIter));
|
||||
|
||||
// Negate the constant.
|
||||
auto NegConstant = IREmit->_Constant(-Constant2);
|
||||
|
||||
// Replace the second source with the negated constant.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUB: {
|
||||
@@ -690,6 +599,21 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
uint64_t Constant1, Constant2;
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(IROp->Args[1], &Constant2) &&
|
||||
Op->Shift == IR::ShiftType::LSL) {
|
||||
// Optimize the LSL case when we know both sources are constant.
|
||||
// This is a pattern that shows up with direction flag calculations if DF was set just before the operation.
|
||||
uint64_t NewConstant = (Constant1 - (Constant2 << Op->ShiftAmount)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_AND: {
|
||||
auto Op = IROp->CW<IR::IROp_And>();
|
||||
uint64_t Constant1{};
|
||||
@@ -721,6 +645,8 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
/* TODO: restore this when we have rmif or something? */
|
||||
#if 0
|
||||
case OP_TESTNZ: {
|
||||
auto Op = IROp->CW<IR::IROp_TestNZ>();
|
||||
uint64_t Constant1{};
|
||||
@@ -735,6 +661,7 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
case OP_OR: {
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
uint64_t Constant1{};
|
||||
@@ -906,12 +833,17 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t SourceMask =
|
||||
Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IROp->Size * 8;
|
||||
uint64_t DestMask =
|
||||
DestSizeInBits == 64 ? ~0ULL : ((1ULL << DestSizeInBits) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
NewConstant <<= 64 - Op->Width;
|
||||
NewConstant >>= 64 - Op->Width;
|
||||
NewConstant &= DestMask;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
|
||||
Changed = true;
|
||||
@@ -993,37 +925,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
auto Select = IREmit->GetOpHeader(Op->Header.Args[0]);
|
||||
|
||||
uint64_t Constant;
|
||||
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
|
||||
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
|
||||
|
||||
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
|
||||
if (SelectCmpClass == GPRPairClass) {
|
||||
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
|
||||
break;
|
||||
}
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Select->Args[2], &Constant1) && IREmit->IsValueConstant(Select->Args[3], &Constant2)) {
|
||||
if (Constant1 == 1 && Constant2 == 0) {
|
||||
auto slc = Select->C<IR::IROp_Select>();
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(Select->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(Select->Args[1]));
|
||||
Op->Cond = slc->Cond;
|
||||
Op->CompareSize = slc->CompareSize;
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1064,20 +965,24 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
case OP_SUB:
|
||||
case OP_ADDNZCV:
|
||||
case OP_SUBNZCV:
|
||||
case OP_ADDWITHFLAGS:
|
||||
case OP_SUBWITHFLAGS:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
if (IsImmAddSub(Constant2) && Op->Header.Size >= 4) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_SUBNZCV) {
|
||||
// If the first source is zero, we can use a NEGS instruction.
|
||||
} else if (IROp->Op == OP_SUBNZCV || IROp->Op == OP_SUBWITHFLAGS || IROp->Op == OP_SUB) {
|
||||
// TODO: Generalize this
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
@@ -1090,16 +995,70 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (IsImmLogical(Constant1, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
|
||||
bool Bitwise = Op->Cond == COND_ANDZ ||
|
||||
Op->Cond == COND_ANDNZ;
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (Bitwise ? IsImmLogical(Constant1, IROp->Size * 8) : IsImmAddSub(Constant1)) {
|
||||
if (IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
@@ -1130,6 +1089,33 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
uint64_t Constant0{};
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1) &&
|
||||
Constant1 == 0)
|
||||
{
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant0) &&
|
||||
(Constant0 == 1 || Constant0 == AllOnes))
|
||||
{
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
@@ -1173,6 +1159,7 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_AND:
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
@@ -1257,6 +1244,35 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1276,8 +1292,6 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
CodeMotionAroundSelects(IREmit, CurrentIR);
|
||||
FCMPOptimization(IREmit, CurrentIR);
|
||||
LoadMemStoreMemImmediatePooling(IREmit, CurrentIR);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
@@ -5,10 +5,10 @@ tags: ir|opts
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
@@ -26,7 +26,7 @@ private:
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
bool Changed = false;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
|
||||
@@ -41,28 +41,71 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
|
||||
if (IROp->Op == OP_SYSCALL ||
|
||||
IROp->Op == OP_INLINESYSCALL) {
|
||||
FEXCore::IR::SyscallFlags Flags{};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
switch (IROp->Op) {
|
||||
case OP_SYSCALL:
|
||||
case OP_INLINESYSCALL: {
|
||||
FEXCore::IR::SyscallFlags Flags{};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ATOMICFETCHADD:
|
||||
case OP_ATOMICFETCHSUB:
|
||||
case OP_ATOMICFETCHAND:
|
||||
case OP_ATOMICFETCHCLR:
|
||||
case OP_ATOMICFETCHOR:
|
||||
case OP_ATOMICFETCHXOR:
|
||||
case OP_ATOMICFETCHNEG: {
|
||||
// If the result of the atomic fetch is completely unused, convert it to a non-fetching atomic operation.
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
switch (IROp->Op) {
|
||||
case OP_ATOMICFETCHADD:
|
||||
IROp->Op = OP_ATOMICADD;
|
||||
break;
|
||||
case OP_ATOMICFETCHSUB:
|
||||
IROp->Op = OP_ATOMICSUB;
|
||||
break;
|
||||
case OP_ATOMICFETCHAND:
|
||||
IROp->Op = OP_ATOMICAND;
|
||||
break;
|
||||
case OP_ATOMICFETCHCLR:
|
||||
IROp->Op = OP_ATOMICCLR;
|
||||
break;
|
||||
case OP_ATOMICFETCHOR:
|
||||
IROp->Op = OP_ATOMICOR;
|
||||
break;
|
||||
case OP_ATOMICFETCHXOR:
|
||||
IROp->Op = OP_ATOMICXOR;
|
||||
break;
|
||||
case OP_ATOMICFETCHNEG:
|
||||
IROp->Op = OP_ATOMICNEG;
|
||||
break;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
|
||||
// Skip over anything that has side effects
|
||||
// Use count tracking can't safely remove anything with side effects
|
||||
if (!HasSideEffects) {
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
NumRemoved++;
|
||||
Changed = true;
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
@@ -74,7 +117,7 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
return NumRemoved != 0;
|
||||
return Changed;
|
||||
}
|
||||
|
||||
void DeadCodeElimination::markUsed(OrderedNodeWrapper *CodeOp, IROp_Header *IROp) {
|
||||
|
||||
@@ -6,12 +6,12 @@ desc: Transforms ContextLoad/Store to temporaries, similar to mem2reg
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -277,6 +277,24 @@ namespace {
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, pf_raw),
|
||||
sizeof(FEXCore::Core::CPUState::pf_raw),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, af_raw),
|
||||
sizeof(FEXCore::Core::CPUState::af_raw),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
@@ -419,6 +437,10 @@ namespace {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// PF/AF
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
@@ -522,7 +544,8 @@ bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *Loc
|
||||
|
||||
bool RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode) {
|
||||
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
|
||||
Info = RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode, CodeNode);
|
||||
RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode,
|
||||
CodeNode);
|
||||
// TODO: Optimize redundant stores.
|
||||
// ContextMemberInfo PreviousMemberInfoCopy = *Info;
|
||||
return false;
|
||||
|
||||
@@ -6,11 +6,11 @@ desc: Cross block store-after-store elimination
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
@@ -6,11 +6,11 @@ desc: Sorts the ssa storage in memory, needed for RA and others
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
@@ -6,6 +6,8 @@ desc: Prints IR
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
@@ -6,12 +6,13 @@ desc: Sanity checking pass
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/IRValidation.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
Loaded 100 of 528 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user