mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 17:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2f5ebf1dd1 | ||
|
|
8f157e45bb | ||
|
|
6b259e2731 | ||
|
|
200660aba5 | ||
|
|
065c12cfbb | ||
|
|
318972620f | ||
|
|
e7f54d1592 | ||
|
|
a8571282b2 | ||
|
|
fc28062052 | ||
|
|
cc6306aa32 | ||
|
|
f0caa81253 | ||
|
|
cd98871f8f | ||
|
|
66e0d46d89 | ||
|
|
1dd5642e46 | ||
|
|
9ca34ca306 | ||
|
|
69f39a0bc3 | ||
|
|
50cf74db0a | ||
|
|
6886d8ff65 | ||
|
|
c1d118c1d4 | ||
|
|
5e46d63c42 | ||
|
|
863a59a8e2 | ||
|
|
c37fcf136a | ||
|
|
02a2292115 | ||
|
|
a483bc9837 | ||
|
|
120a6b85f4 | ||
|
|
9fea774a93 | ||
|
|
bf1e619ead | ||
|
|
0f8fcfc43e | ||
|
|
698b7fda06 | ||
|
|
23caa6e20f | ||
|
|
34e39c996e | ||
|
|
16ed20cfae | ||
|
|
ef368ceafa | ||
|
|
45480ef32c | ||
|
|
4de69029e5 | ||
|
|
c065770f48 | ||
|
|
27957ea051 | ||
|
|
94e9d1ab3b | ||
|
|
a374a9af35 | ||
|
|
3e80416eb6 | ||
|
|
b35c6c6d22 | ||
|
|
69d26cfee6 | ||
|
|
e3be1540f1 | ||
|
|
8e2b0d10e5 | ||
|
|
57c5761920 | ||
|
|
0841ff5feb | ||
|
|
45115384b5 | ||
|
|
dae2563850 | ||
|
|
fb4df5a0b7 | ||
|
|
b2f0303d1e | ||
|
|
f8b2a0b4d8 | ||
|
|
e1fcb78ce3 | ||
|
|
d6309088c0 | ||
|
|
1fb3a2e28f | ||
|
|
df25d4e03e | ||
|
|
9c2f0287e0 | ||
|
|
9912d41714 | ||
|
|
8b2cd87d9e | ||
|
|
3e6d23ae7e | ||
|
|
c96c39d5b1 | ||
|
|
a4556e90cd | ||
|
|
4d2c4b4423 | ||
|
|
530de3f031 | ||
|
|
c9622f6fd4 | ||
|
|
854628d959 | ||
|
|
35dbf6c44b | ||
|
|
b9fec7436f | ||
|
|
808d19c374 | ||
|
|
441d7205ed | ||
|
|
8b3d3b68c6 | ||
|
|
aa0e038ef7 | ||
|
|
5e5e5a35d9 | ||
|
|
10e35a55ea | ||
|
|
83cebea780 | ||
|
|
91bbb92c50 | ||
|
|
c7dd6ff28a | ||
|
|
df761a99ce | ||
|
|
359416e2b6 | ||
|
|
e140c0d60c | ||
|
|
0030971f6f | ||
|
|
3a90aaf1e6 | ||
|
|
f8a199af49 | ||
|
|
3a5de8e10c | ||
|
|
3c881809f4 | ||
|
|
8b6e9e08c0 | ||
|
|
3c8da3e3b4 | ||
|
|
0a39d909b2 | ||
|
|
d35d1092a4 | ||
|
|
7ae655c56a | ||
|
|
58f35ba413 | ||
|
|
e61eb24ec2 | ||
|
|
2e93d2ce51 | ||
|
|
9d21e1efd5 | ||
|
|
e0e6b3ad6b | ||
|
|
815cdc5b3c | ||
|
|
bc20f1e684 | ||
|
|
20e5f2bec6 | ||
|
|
c3b6fa55b6 | ||
|
|
69045db3a9 | ||
|
|
7b2240c80b | ||
|
|
dcfbd90dd7 | ||
|
|
70e6ab5782 | ||
|
|
175879823f | ||
|
|
2271a90adb | ||
|
|
2bdde5845e | ||
|
|
0c40497a01 | ||
|
|
45dff0f550 | ||
|
|
e9035ef6ee | ||
|
|
02ca94e6e6 | ||
|
|
432b7d2dc8 | ||
|
|
181d315d2c | ||
|
|
d5f7e616eb | ||
|
|
9a8869e8a4 | ||
|
|
85d56ed76f | ||
|
|
06c827b5c8 | ||
|
|
41259ff361 | ||
|
|
cccee1a668 | ||
|
|
c46b35362b | ||
|
|
2f96a6d8bf | ||
|
|
c78a47a3e5 | ||
|
|
b049721683 | ||
|
|
11c06fe9fe | ||
|
|
c5f6e53d0d | ||
|
|
1184672bb1 | ||
|
|
57d3a2ba35 | ||
|
|
1d813b0183 | ||
|
|
9b24931518 | ||
|
|
7e5b8b7bdf | ||
|
|
7e36473aff | ||
|
|
adbb512306 | ||
|
|
b951f4ad4b | ||
|
|
48900662ae | ||
|
|
da31a66c07 | ||
|
|
ca710b1cbb | ||
|
|
a4d7eec145 | ||
|
|
232c2fe87f | ||
|
|
661112cfd4 | ||
|
|
d4a84eaa9a | ||
|
|
99f8af64d0 | ||
|
|
1541ea9ffc | ||
|
|
ced86e693c | ||
|
|
53920d5bd3 | ||
|
|
15c5a9dac0 | ||
|
|
d63cfdbe7d | ||
|
|
569461a01c | ||
|
|
37c7dee236 | ||
|
|
cc230091c8 | ||
|
|
bac33cf246 | ||
|
|
7a9c0506b4 | ||
|
|
fbd7c15a4b | ||
|
|
64edf24bc7 | ||
|
|
3f456d683f | ||
|
|
aac0824fd3 | ||
|
|
1b32ca0b93 | ||
|
|
c29456aac9 | ||
|
|
715f25d059 | ||
|
|
c85e31ec7a | ||
|
|
0d6bfc4fa4 | ||
|
|
106917add2 | ||
|
|
1e10a2bac3 | ||
|
|
63dd09cd6f | ||
|
|
c7e6935f42 | ||
|
|
1be5054e86 | ||
|
|
f40755aca7 | ||
|
|
d49b78cf34 | ||
|
|
10e80ae064 | ||
|
|
f3ebb214aa | ||
|
|
dc41c3bd9f | ||
|
|
56ff09f3ac | ||
|
|
1707b27d14 | ||
|
|
7c92963eca | ||
|
|
ecf82c90ee | ||
|
|
580f06fe00 | ||
|
|
5a403b7765 | ||
|
|
f7367e56af | ||
|
|
f066abc151 | ||
|
|
2bee92f332 | ||
|
|
7d9ed4e1bf | ||
|
|
24696e6b98 | ||
|
|
71f658b07d | ||
|
|
920f56353f | ||
|
|
e0fe9167ea | ||
|
|
45a2349a0d | ||
|
|
d7b0e8469e | ||
|
|
8afc3b8e23 | ||
|
|
9caa63d5b2 | ||
|
|
1d32df91ce | ||
|
|
fa973f65bf | ||
|
|
c027f02e5a | ||
|
|
9cee0126d7 | ||
|
|
c713602d56 | ||
|
|
c8293cbbda | ||
|
|
a9c51388cf | ||
|
|
ad1d65e91a | ||
|
|
3a6c7803e8 | ||
|
|
aa837eddd5 | ||
|
|
faef57838f | ||
|
|
04d4c5e017 | ||
|
|
9158877569 | ||
|
|
1eae07f1b8 | ||
|
|
14b22487f1 | ||
|
|
1ac7cd5835 | ||
|
|
5336f01725 | ||
|
|
b8b66b1829 | ||
|
|
fd3e988a20 | ||
|
|
b1d98f4e58 | ||
|
|
9e7daf61d0 | ||
|
|
6fbe25753b | ||
|
|
03f0edc5b5 | ||
|
|
5536f1e835 | ||
|
|
0de36706da | ||
|
|
17722dad6d | ||
|
|
0371599996 | ||
|
|
199649b30f | ||
|
|
4ef35488db | ||
|
|
70a91ee6ce | ||
|
|
418a27e47e | ||
|
|
61c76d02cc | ||
|
|
d98641221d | ||
|
|
8a14f87a44 | ||
|
|
02ce71734c | ||
|
|
96c2743280 | ||
|
|
7bfc34b51c | ||
|
|
40d820fd05 | ||
|
|
d69287aaf7 | ||
|
|
d475b0ba9e | ||
|
|
1638b744b7 | ||
|
|
8b19894a06 | ||
|
|
d2e0dc99de | ||
|
|
d04e40b5fd | ||
|
|
75d797b5cd | ||
|
|
ecf4891087 | ||
|
|
0e1a418678 | ||
|
|
5bef13df94 | ||
|
|
d8386121a8 | ||
|
|
000677abb6 | ||
|
|
64eb87e9b5 | ||
|
|
aa5e92bee2 | ||
|
|
0bf79dc5d6 | ||
|
|
adb2171c0a | ||
|
|
d6f8923f86 | ||
|
|
cf91ab9d5f | ||
|
|
a0fb9531db | ||
|
|
eca9353b28 | ||
|
|
a259730639 | ||
|
|
2e93d10eba | ||
|
|
70a3ceb64e | ||
|
|
b726f60afd | ||
|
|
2fa1a64999 | ||
|
|
a42b659af9 | ||
|
|
004c3230a4 | ||
|
|
2332c41510 | ||
|
|
ec3039c5a2 | ||
|
|
639d6e6071 | ||
|
|
cd518d4726 | ||
|
|
b7d9c00dff | ||
|
|
4b17575f5a | ||
|
|
5ba4bba138 | ||
|
|
b3ee5dba0f | ||
|
|
d87ff5afa9 | ||
|
|
62a24bd38f | ||
|
|
8d373c15b8 | ||
|
|
671f3e74a4 | ||
|
|
7e810233d9 | ||
|
|
4700dbd676 | ||
|
|
74e18f4317 | ||
|
|
1eea95cf18 | ||
|
|
7291b10727 | ||
|
|
6804916697 | ||
|
|
819e61bf14 | ||
|
|
b8f7e4c8ec | ||
|
|
17bcc0eed4 | ||
|
|
13003da289 | ||
|
|
9273538955 | ||
|
|
9750189def | ||
|
|
cb17ee9871 | ||
|
|
4c3b78ba9a | ||
|
|
e00b6a401b | ||
|
|
ac0ab8a7b4 | ||
|
|
0aff3941f4 | ||
|
|
27b022d4d9 | ||
|
|
e188928742 | ||
|
|
780e3c7fb7 | ||
|
|
1b5146d3ac | ||
|
|
80cf3ca6b9 | ||
|
|
2272b30a91 | ||
|
|
b5fb1cb07c | ||
|
|
4ea34a9c22 | ||
|
|
48e7de9f9e | ||
|
|
76dd2369a7 | ||
|
|
78e0cd6e77 | ||
|
|
5514a04cb4 | ||
|
|
99ca78b235 | ||
|
|
ab45db1665 | ||
|
|
bb38bcb67d | ||
|
|
6cc2912542 | ||
|
|
340b2ca624 | ||
|
|
5baa15de03 | ||
|
|
6ddca804d1 | ||
|
|
f9831a85fb | ||
|
|
7261033b7f | ||
|
|
3ad6866198 | ||
|
|
1c7d4165ab | ||
|
|
3e48b1a8ac | ||
|
|
07be100daf | ||
|
|
de9351eefb | ||
|
|
2c44b5b3a1 | ||
|
|
136f1e2fc7 | ||
|
|
ad39add55f | ||
|
|
f3c301e359 | ||
|
|
1c37a1b4d6 | ||
|
|
7222529904 | ||
|
|
f26eccd00f | ||
|
|
73375a76ac | ||
|
|
d4416d200e | ||
|
|
d1b235dd83 | ||
|
|
3ac5e0423a | ||
|
|
fc6de5f3c0 | ||
|
|
b1e475d81d | ||
|
|
4a09a4324f | ||
|
|
47f94327c5 | ||
|
|
a009ed0b6b | ||
|
|
2476a686e7 | ||
|
|
f0db93773f | ||
|
|
fabe824c8b | ||
|
|
78a077397e | ||
|
|
0c4b456aaa | ||
|
|
09185167bc | ||
|
|
5d78c3203c | ||
|
|
fa5322d3f9 | ||
|
|
d21aa5cac2 | ||
|
|
102d5c57cb | ||
|
|
ffb4de9fd9 | ||
|
|
6b3d8886e5 | ||
|
|
ddc10272a0 | ||
|
|
0b5ef00165 | ||
|
|
2a50416fc3 | ||
|
|
ce514d9f83 | ||
|
|
9b77e7fd13 | ||
|
|
abb44d3327 | ||
|
|
11eaf3d48a | ||
|
|
76c2cc2c3e | ||
|
|
0e6c8bd12e | ||
|
|
a9fb008317 | ||
|
|
e9f3a5b3e4 | ||
|
|
ebc45dff45 | ||
|
|
fc4a5ebfd3 | ||
|
|
1d7b688c55 | ||
|
|
f14a5ffbbf | ||
|
|
cada0d593c | ||
|
|
0436540791 | ||
|
|
a87ac86e18 | ||
|
|
1278b23150 | ||
|
|
02f5ea4b9d | ||
|
|
2e14e613d0 | ||
|
|
85c2889652 | ||
|
|
23dd056b60 | ||
|
|
71043e372a | ||
|
|
c412d073b9 | ||
|
|
8fb03ff1b9 | ||
|
|
c2b6aef6f4 | ||
|
|
24547318c6 | ||
|
|
b693112c80 | ||
|
|
48d1184066 | ||
|
|
8da9ebc2e0 | ||
|
|
5ba510474b | ||
|
|
51214d1be1 | ||
|
|
4d6e15d7af | ||
|
|
4721894427 | ||
|
|
d429865b6e | ||
|
|
ca5881a72c | ||
|
|
7151b9daff | ||
|
|
d9b5e28b22 | ||
|
|
aa7954a7d6 | ||
|
|
802c70d1ab | ||
|
|
8f905988e9 | ||
|
|
478c5595ad | ||
|
|
3977e1f29e | ||
|
|
2b1ef97354 | ||
|
|
eaddf7f1a5 | ||
|
|
3237de3085 | ||
|
|
df3d398d31 | ||
|
|
b44b3401b7 | ||
|
|
c28ca0fac9 | ||
|
|
235e2b6c2c | ||
|
|
d68b84bc27 | ||
|
|
edca528608 | ||
|
|
b75e8f2abf | ||
|
|
ec3158e4cd | ||
|
|
9c8c8041e0 | ||
|
|
e3adaacb51 | ||
|
|
c49e11484f | ||
|
|
7b4b9a80fa | ||
|
|
c7ad066987 | ||
|
|
f4d229f1ba | ||
|
|
25a8a00771 | ||
|
|
a67f7422b2 | ||
|
|
ed8150cfb6 | ||
|
|
462a163ba7 | ||
|
|
6374175a64 | ||
|
|
280b15ba2a | ||
|
|
2a0b488e99 | ||
|
|
3ef7c4ab51 | ||
|
|
e72d746036 | ||
|
|
0bb4091e34 | ||
|
|
6822fc595c | ||
|
|
a506a589dd | ||
|
|
684a5977dd | ||
|
|
ac3682e058 | ||
|
|
d5faf01f5a | ||
|
|
ecba1b6838 | ||
|
|
8d8b029285 | ||
|
|
bf6f855868 | ||
|
|
aa6a499329 | ||
|
|
0971650ef9 | ||
|
|
64c4fdccf7 | ||
|
|
d715ffbc8e | ||
|
|
aef801b5b5 | ||
|
|
6c9e29796b | ||
|
|
364bb3ac1e | ||
|
|
428ea68507 | ||
|
|
5cf59408a7 | ||
|
|
1596843015 | ||
|
|
af6582ff5b | ||
|
|
1799d4c675 | ||
|
|
808e1c0330 | ||
|
|
dacd96cab5 | ||
|
|
ca4d3bf64d | ||
|
|
ea38b043c1 | ||
|
|
a39746df2e | ||
|
|
2367a8e50b | ||
|
|
cb121d7f17 | ||
|
|
412793c21d | ||
|
|
ce2286c48b | ||
|
|
d5694d6de0 | ||
|
|
121218aa8a | ||
|
|
6bb53fa758 | ||
|
|
89aa0c5471 | ||
|
|
53fcbf6afa | ||
|
|
2f5643ae6b | ||
|
|
d162ac8b3d | ||
|
|
c9a704fbde | ||
|
|
9d5a822a3a | ||
|
|
50eba4066a | ||
|
|
6116ae5330 | ||
|
|
447226576f | ||
|
|
3f8b872f17 | ||
|
|
e573ddc2db | ||
|
|
fe9aa681f0 | ||
|
|
1b2f2c1559 | ||
|
|
84c75a86c3 | ||
|
|
eedbde6f15 | ||
|
|
4e441e5a08 | ||
|
|
3e287a36c2 | ||
|
|
eadc477695 | ||
|
|
59aa324678 | ||
|
|
219bce1467 | ||
|
|
46bde401bd | ||
|
|
01beac4956 | ||
|
|
fcd981e6b7 | ||
|
|
825833cfcc | ||
|
|
4a4c49bf68 | ||
|
|
763cea423a | ||
|
|
0fee355ff5 | ||
|
|
6f6f3c9dc5 | ||
|
|
25e5d88ab2 | ||
|
|
87013340bb | ||
|
|
47c075ccc9 | ||
|
|
b1a32d4ccf | ||
|
|
7af6a8dbdf | ||
|
|
383e99e4ef | ||
|
|
8c7cfc4d11 | ||
|
|
496ee730c8 | ||
|
|
71f7ff5101 | ||
|
|
1ea00f68a2 | ||
|
|
8df7c2d84f | ||
|
|
46557a7a1f | ||
|
|
d8c2a8271f | ||
|
|
cc4c705fc0 | ||
|
|
22f249fcf6 | ||
|
|
c262362a03 | ||
|
|
212df9aa7b | ||
|
|
691e39ec76 | ||
|
|
107cae2975 | ||
|
|
6742e0c376 | ||
|
|
8f70137b1a | ||
|
|
707db51b1b | ||
|
|
5b5fa1aa29 | ||
|
|
2b9cc9666a | ||
|
|
82eba22292 | ||
|
|
5c84e8f23c | ||
|
|
081b61677a | ||
|
|
9c54814b98 | ||
|
|
35d7b855ed | ||
|
|
0b8799274c | ||
|
|
ace2b737d8 | ||
|
|
ad85268524 | ||
|
|
832a320e22 | ||
|
|
83763df6fd | ||
|
|
bee868e9ba | ||
|
|
c41de81694 | ||
|
|
41aaeb1ff0 | ||
|
|
16be2792ab | ||
|
|
6610bb355c | ||
|
|
4312fd7291 | ||
|
|
89a225a96d | ||
|
|
a590977639 | ||
|
|
0d0d116bde | ||
|
|
341bdb5a54 | ||
|
|
2f3dbfb289 | ||
|
|
169cfbbeed | ||
|
|
f97a4afd8f | ||
|
|
868e4a6d81 | ||
|
|
6f48f7d3ac | ||
|
|
1ed3ecb409 | ||
|
|
2cb455b9d4 | ||
|
|
d4b5bf0f78 | ||
|
|
69013772c1 | ||
|
|
3448c83431 | ||
|
|
e4c84542ea | ||
|
|
acddc0323b | ||
|
|
c4285f0d30 | ||
|
|
977d6dd247 | ||
|
|
2bb27fffb7 | ||
|
|
0261ed353d | ||
|
|
96fecfd7c5 | ||
|
|
9fac1b8105 | ||
|
|
95fbcd7b9a | ||
|
|
6adf227611 | ||
|
|
b5cb429243 | ||
|
|
26ba8079a3 | ||
|
|
f999d30bc5 | ||
|
|
ecb1cc4ed4 | ||
|
|
5622bcae16 | ||
|
|
f4539ee289 | ||
|
|
d2138694b4 | ||
|
|
1767e21273 | ||
|
|
8f9d799342 | ||
|
|
a3b0b246f4 | ||
|
|
dee85f14fe | ||
|
|
27309114be | ||
|
|
704afed97b | ||
|
|
121f0a2c6c | ||
|
|
1c580ec92c | ||
|
|
ab8fc721a0 | ||
|
|
790447115c | ||
|
|
80abeac28a | ||
|
|
54915f87ce | ||
|
|
f34f1309a7 | ||
|
|
0ad52b7d19 | ||
|
|
809f60df06 | ||
|
|
cbdcd8253c | ||
|
|
7ac2cd7cc8 | ||
|
|
cf1bb1348c | ||
|
|
b60a26ff9e | ||
|
|
b805c07342 | ||
|
|
8d69f539ac | ||
|
|
c8c0054f67 | ||
|
|
1085385bbe | ||
|
|
56460b220c | ||
|
|
a583ebe590 | ||
|
|
ce8175a800 | ||
|
|
c987e1ef44 | ||
|
|
b36ec152d2 | ||
|
|
44c62e703e | ||
|
|
0f59c1d5e3 | ||
|
|
5739f0b459 | ||
|
|
4145fabfb6 |
No files matched your search
@@ -64,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -188,6 +188,40 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Install
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: Test GL No-Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
|
||||
|
||||
- name: No thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
|
||||
|
||||
- name: Test GL Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
|
||||
|
||||
- name: Thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
@@ -207,4 +241,3 @@ jobs:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
name: Vixl Simulator run
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
# Only the x86-64 runner is fast enough to run this
|
||||
arch: [[self-hosted, x64], [self-hosted, ARMv8.4]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"ThunksDB": {
|
||||
"GL": 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"ThunksDB": {
|
||||
"Vulkan": 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
if(NOT EXISTS "@CMAKE_BINARY_DIR@/install_manifest.txt")
|
||||
message(FATAL_ERROR "Cannot find install manifest: @CMAKE_BINARY_DIR@/install_manifest.txt")
|
||||
endif()
|
||||
|
||||
file(READ "@CMAKE_BINARY_DIR@/install_manifest.txt" files)
|
||||
string(REGEX REPLACE "\n" ";" files "${files}")
|
||||
foreach(file ${files})
|
||||
message(STATUS "Uninstalling $ENV{DESTDIR}${file}")
|
||||
if(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
|
||||
exec_program(
|
||||
"@CMAKE_COMMAND@" ARGS "-E remove \"$ENV{DESTDIR}${file}\""
|
||||
OUTPUT_VARIABLE rm_out
|
||||
RETURN_VALUE rm_retval
|
||||
)
|
||||
if(NOT "${rm_retval}" STREQUAL 0)
|
||||
message(FATAL_ERROR "Problem when removing $ENV{DESTDIR}${file}")
|
||||
endif()
|
||||
else(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
|
||||
message(STATUS "File $ENV{DESTDIR}${file} does not exist.")
|
||||
endif()
|
||||
endforeach()
|
||||
+72
-6
@@ -7,6 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -27,10 +28,36 @@ option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CMakeFiles/cmake_uninstall.cmake.in"
|
||||
"${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake"
|
||||
IMMEDIATE @ONLY)
|
||||
|
||||
add_custom_target(uninstall
|
||||
COMMAND ${CMAKE_COMMAND} -P ${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake)
|
||||
endif()
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
@@ -84,10 +111,9 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
@@ -363,8 +389,6 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
include_directories(FEXHeaderUtils/)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
@@ -403,8 +427,28 @@ if (BUILD_THUNKS)
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DBITNESS=64"
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs-32
|
||||
PREFIX guest-libs-32
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest_32"
|
||||
CMAKE_ARGS
|
||||
"-DBITNESS=32"
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
@@ -422,6 +466,28 @@ if (BUILD_THUNKS)
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs-32
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_guest-libs)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
|
||||
@@ -165,6 +165,14 @@
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
Vendored
+10
-4
@@ -9,12 +9,19 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
endif()
|
||||
|
||||
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
|
||||
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# If the vixl simulator is enabled then we are using the ARM64 JIT
|
||||
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
|
||||
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
|
||||
else()
|
||||
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
|
||||
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
|
||||
endif()
|
||||
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
@@ -27,7 +34,6 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
|
||||
include(CheckCXXCompilerFlag)
|
||||
include(CheckIncludeFileCXX)
|
||||
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
# Useful to have for freestanding libFEXCore
|
||||
add_subdirectory(External/vixl/)
|
||||
|
||||
+10
-5
@@ -135,7 +135,6 @@ set (SRCS
|
||||
Interface/IR/Passes/PhiValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
@@ -143,6 +142,7 @@ set (SRCS
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
@@ -177,6 +177,11 @@ if (_M_ARM_64)
|
||||
list(APPEND DEFINES -D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# We can run the simulator on both x86-64 or AArch64 hosts
|
||||
list(APPEND DEFINES -DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JIT_X86_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/JIT/x86_64/JIT.cpp
|
||||
@@ -213,7 +218,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
@@ -359,14 +364,14 @@ endfunction()
|
||||
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base fmt::fmt tiny-json)
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
@@ -374,7 +379,7 @@ endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
+10
-6
@@ -211,10 +211,12 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
|
||||
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
|
||||
@@ -629,7 +631,7 @@ namespace JSON {
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, bool Global);
|
||||
explicit AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
@@ -681,8 +683,10 @@ namespace JSON {
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, bool Global)
|
||||
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
|
||||
AppLoader::AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
@@ -754,8 +758,8 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
|
||||
+10
-2
@@ -16,10 +16,11 @@
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Default": "false",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation"
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
]
|
||||
},
|
||||
"MaxInst": {
|
||||
@@ -90,6 +91,13 @@
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkGuestLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
|
||||
+2
-1
@@ -15,6 +15,7 @@
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
@@ -189,7 +190,7 @@ namespace FEXCore::Context {
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
|
||||
@@ -17,23 +17,35 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
: vixl::aarch64::Assembler(size ? (byte*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : reinterpret_cast<byte*>(~0ULL),
|
||||
size,
|
||||
vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
if (ctx->HostFeatures.SupportsAtomics) {
|
||||
// Hypervisor can hide this on the c630?
|
||||
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
|
||||
}
|
||||
#endif
|
||||
|
||||
SetCPUFeatures(Features);
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto CodeBuffer = GetBuffer();
|
||||
if (CodeBuffer->GetCapacity()) {
|
||||
FEXCore::Allocator::munmap(CodeBuffer->GetStartAddress<void*>(), CodeBuffer->GetCapacity());
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = Reg.IsX();
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
@@ -214,7 +226,8 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
mov(TMP4, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
st1b(Reg.Z().VnB(), PRED_TMP_32B, SVEMemOperand(STATE, TMP4));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -257,11 +270,19 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
ptrue(PRED_TMP_16B.VnB(), SVE_VL16);
|
||||
ptrue(PRED_TMP_32B.VnB(), SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
mov(TMP4, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b(Reg.Z().VnB(), PRED_TMP_32B.Zeroing(), SVEMemOperand(STATE, TMP4));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -286,20 +307,31 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR() {
|
||||
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (RA64.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(sp, sp, SPOffset);
|
||||
int i = 0;
|
||||
|
||||
for (auto RA : RAFPR)
|
||||
{
|
||||
str(RA.Q(), MemOperand(sp, i * 8));
|
||||
i+=2;
|
||||
if (CanUseSVE) {
|
||||
for (const auto& RA : RAFPR) {
|
||||
mov(TMP4, i * 8);
|
||||
st1b(RA.Z().VnB(), PRED_TMP_32B, SVEMemOperand(sp, TMP4));
|
||||
i += 4;
|
||||
}
|
||||
} else {
|
||||
for (const auto& RA : RAFPR) {
|
||||
str(RA.Q(), MemOperand(sp, i * 8));
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
|
||||
#if 0 // All GPRs should be caller saved
|
||||
for (auto RA : RA64)
|
||||
{
|
||||
for (const auto& RA : RA64) {
|
||||
str(RA, MemOperand(sp, i * 8));
|
||||
i++;
|
||||
}
|
||||
@@ -309,18 +341,29 @@ void Arm64Emitter::PushDynamicRegsAndLR() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (RA64.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
int i = 0;
|
||||
|
||||
for (auto RA : RAFPR)
|
||||
{
|
||||
ldr(RA.Q(), MemOperand(sp, i * 8));
|
||||
i+=2;
|
||||
if (CanUseSVE) {
|
||||
for (const auto& RA : RAFPR) {
|
||||
mov(TMP4, i * 8);
|
||||
ld1b(RA.Z().VnB(), PRED_TMP_32B.Zeroing(), SVEMemOperand(sp, TMP4));
|
||||
i += 4;
|
||||
}
|
||||
} else {
|
||||
for (const auto& RA : RAFPR) {
|
||||
ldr(RA.Q(), MemOperand(sp, i * 8));
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
|
||||
#if 0 // All GPRs should be caller saved
|
||||
for (auto RA : RA64)
|
||||
{
|
||||
for (const auto& RA : RA64) {
|
||||
ldr(RA, MemOperand(sp, i * 8));
|
||||
i++;
|
||||
}
|
||||
|
||||
@@ -8,6 +8,10 @@
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
@@ -58,15 +62,41 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
v8, v9, v10, v11, v12, v13, v14, v15
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
#define STATE x28
|
||||
|
||||
// GPR temporaries. Only x3 can be used across spill boundaries
|
||||
// so if these ever need to change, be very careful about that.
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
#define TMP3 x2
|
||||
#define TMP4 x3
|
||||
|
||||
// Vector temporaries
|
||||
#define VTMP1 v1
|
||||
#define VTMP2 v2
|
||||
#define VTMP3 v3
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
#define PRED_TMP_16B p6
|
||||
#define PRED_TMP_32B p7
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
@@ -83,6 +113,71 @@ protected:
|
||||
void PopCalleeSavedRegisters();
|
||||
|
||||
void Align16B();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Generates a vixl simulator runtime call.
|
||||
//
|
||||
// This matches behaviour of vixl's macro assembler, but we need to reimplement it since we aren't using the macro assembler.
|
||||
// This isn't too complex with how vixl emits this.
|
||||
//
|
||||
// Emit:
|
||||
// 1) hlt(kRuntimeCallOpcode)
|
||||
// 2) Simulator wrapper handler
|
||||
// 3) Function to call
|
||||
// 4) Style of the function call (Call versus tail-call)
|
||||
template<typename R, typename... P>
|
||||
void GenerateRuntimeCall(R (*Function)(P...)) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
|
||||
|
||||
hlt(kRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc(SimulatorWrapperAddress);
|
||||
|
||||
// Runtime function address to call
|
||||
dc(FunctionAddress);
|
||||
|
||||
// Call type
|
||||
dc32(kCallRuntime);
|
||||
}
|
||||
|
||||
template<typename R, typename... P>
|
||||
void GenerateIndirectRuntimeCall(vixl::aarch64::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
hlt(kIndirectRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc(SimulatorWrapperAddress);
|
||||
|
||||
// Register that contains the function to call
|
||||
dc(Reg.GetCode());
|
||||
|
||||
// Call type
|
||||
dc32(kCallRuntime);
|
||||
}
|
||||
|
||||
template<>
|
||||
void GenerateIndirectRuntimeCall<float, __uint128_t>(vixl::aarch64::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
|
||||
|
||||
hlt(kIndirectRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc(SimulatorWrapperAddress);
|
||||
|
||||
// Register that contains the function to call
|
||||
dc(Reg.GetCode());
|
||||
|
||||
// Call type
|
||||
dc32(kCallRuntime);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
+1
-1
@@ -421,7 +421,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(1 << 1) | // PCLMULQDQ
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(0 << 4) | // DS-CPL
|
||||
|
||||
+19
-15
@@ -44,6 +44,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
@@ -221,7 +222,7 @@ namespace FEXCore::Context {
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
|
||||
#else
|
||||
@@ -238,16 +239,16 @@ namespace FEXCore::Context {
|
||||
|
||||
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
|
||||
|
||||
#if (_M_X86_64)
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#elif (_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateArm64(this, DispatcherConfig);
|
||||
#elif JIT_X86_64
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
|
||||
#endif
|
||||
|
||||
// Initialize common signal handlers
|
||||
|
||||
|
||||
auto PauseHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
|
||||
};
|
||||
@@ -573,7 +574,7 @@ namespace FEXCore::Context {
|
||||
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, Thread);
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
@@ -674,6 +675,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
@@ -740,7 +743,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
@@ -749,7 +754,7 @@ namespace FEXCore::Context {
|
||||
|
||||
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
|
||||
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
if (Handler != CustomIRHandlers.end()) {
|
||||
TotalInstructions = 1;
|
||||
@@ -872,7 +877,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
@@ -1011,6 +1016,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
@@ -1182,7 +1188,7 @@ namespace FEXCore::Context {
|
||||
|
||||
static void InvalidateGuestCodeRangeInternal(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(CTX->ThreadCreationMutex);
|
||||
|
||||
|
||||
for (auto &Thread : CTX->Threads) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
@@ -1190,7 +1196,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
}
|
||||
|
||||
@@ -1206,8 +1212,6 @@ namespace FEXCore::Context {
|
||||
IsMemoryShared = true;
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
LogMan::Msg::IFmt("Migrating to shared memory mode");
|
||||
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
LogMan::Throw::AFmt(Threads.size() == 1, "First MarkMemoryShared called must be before creating any threads");
|
||||
|
||||
@@ -1233,9 +1237,9 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(Thread->CTX->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->DebugStore.erase(GuestRIP);
|
||||
|
||||
@@ -38,11 +38,20 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 8192;
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&Decoder}
|
||||
#endif
|
||||
{
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode a 256-bit vector width if we are running in the simulator.
|
||||
Simulator.SetVectorLengthInBits(256);
|
||||
#endif
|
||||
|
||||
SetAllowAssembler(true);
|
||||
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
@@ -178,7 +187,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
@@ -206,8 +220,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
blr(x3);
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
@@ -266,8 +284,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ldr(x3, &l_CompileBlock);
|
||||
|
||||
// X2 contains our guest RIP
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(x3);
|
||||
#else
|
||||
blr(x3); // { CTX, Frame, RIP}
|
||||
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
@@ -349,7 +370,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x2, &l_Sleep);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void *, void *>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PauseReturnInstruction = GetCursorAddress<uint64_t>();
|
||||
// Fault to start running again
|
||||
@@ -412,11 +437,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -431,11 +459,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -450,11 +481,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -469,11 +503,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -504,13 +541,27 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void Arm64Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.RunFrom(reinterpret_cast<Instruction const*>(DispatchPtr));
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.WriteXRegister(1, RIP);
|
||||
Simulator.RunFrom(reinterpret_cast<Instruction const*>(CallbackPtr));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
@@ -546,7 +597,7 @@ size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t Gues
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
@@ -594,7 +645,7 @@ void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -20,6 +24,11 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) override;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) override;
|
||||
#endif
|
||||
|
||||
protected:
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
@@ -29,6 +38,11 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
vixl::aarch64::Simulator Simulator;
|
||||
#endif
|
||||
};
|
||||
|
||||
}
|
||||
@@ -212,12 +212,20 @@ void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread,
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
@@ -565,10 +573,13 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
@@ -581,10 +592,8 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
|
||||
@@ -32,7 +32,7 @@ struct DispatcherConfig {
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -75,12 +75,12 @@ public:
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
|
||||
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
virtual void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
|
||||
+19
-4
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
@@ -450,8 +451,13 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DestSize = 2;
|
||||
}
|
||||
else if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_128BIT);
|
||||
DestSize = 16;
|
||||
if (Options.L) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_256BIT);
|
||||
DestSize = 32;
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_128BIT);
|
||||
DestSize = 16;
|
||||
}
|
||||
}
|
||||
else if (HasNarrowingDisplacement &&
|
||||
(DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_DEF ||
|
||||
@@ -482,7 +488,14 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
|
||||
if (Options.L) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
|
||||
}
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_256BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
|
||||
}
|
||||
else if (HasNarrowingDisplacement &&
|
||||
(SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_DEF ||
|
||||
@@ -776,6 +789,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
if (Op == 0xC5) { // Two byte VEX
|
||||
pp = Byte1 & 0b11;
|
||||
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
|
||||
options.L = (Byte1 & 0b100) != 0;
|
||||
}
|
||||
else { // 0xC4 = Three byte VEX
|
||||
const uint8_t Byte2 = ReadByte();
|
||||
@@ -783,6 +797,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
map_select = Byte1 & 0b11111;
|
||||
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
options.L = (Byte2 & 0b100) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
@@ -1132,6 +1147,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1166,7 +1182,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
|
||||
@@ -49,6 +49,7 @@ private:
|
||||
struct DecodedHeader {
|
||||
uint8_t vvvv; // Encoded operand in a VEX prefix.
|
||||
bool w; // VEX.W bit.
|
||||
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
|
||||
};
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
+52
-33
@@ -1,7 +1,7 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
@@ -50,8 +50,12 @@ static uint32_t GetDCZID() {
|
||||
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
@@ -61,15 +65,26 @@ HostFeatures::HostFeatures() {
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
SupportsAVX = true;
|
||||
#else
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
@@ -79,37 +94,6 @@ HostFeatures::HostFeatures() {
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
#endif
|
||||
#ifdef _M_X86_64
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
@@ -130,6 +114,40 @@ HostFeatures::HostFeatures() {
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
#endif
|
||||
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
SupportsPMULL_128Bit = Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
@@ -139,5 +157,6 @@ HostFeatures::HostFeatures() {
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
+35
-17
@@ -894,33 +894,51 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Shift = ElementSizeBits * Op->Index;
|
||||
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
|
||||
"OpSize is too large for VExtractToGPR: {}", OpSize);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
if (SourceSize >= SSERegSize) {
|
||||
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
|
||||
|
||||
const auto GetResult = [&] {
|
||||
if (Shift >= SSEBitSize) {
|
||||
const auto NormalizedShift = Shift - SSEBitSize;
|
||||
return (Src.Upper >> NormalizedShift) & SourceMask;
|
||||
} else {
|
||||
return (Src.Lower >> Shift) & SourceMask;
|
||||
}
|
||||
};
|
||||
|
||||
const auto Result = GetResult();
|
||||
memcpy(GDP, &Result, ElementSize);
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
const uint64_t Result = (Src >> Shift) & SourceMask;
|
||||
GD = Result;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -13,22 +13,46 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
|
||||
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
const auto Scalar = Src2 & Mask;
|
||||
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
|
||||
: Offset;
|
||||
|
||||
// Now shift into place and set all bits but
|
||||
// the ones where we're going to insert our value.
|
||||
Mask <<= ScaledOffset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
Dst |= Src2 << Offset;
|
||||
|
||||
const auto Dst = [&] {
|
||||
if (InUpperLane) {
|
||||
return InterpVector256{
|
||||
.Lower = Src1.Lower,
|
||||
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
|
||||
};
|
||||
} else {
|
||||
return InterpVector256{
|
||||
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
|
||||
.Upper = Src1.Upper,
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
@@ -89,63 +113,73 @@ DEF_OP(Vector_SToF) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
@@ -165,19 +199,22 @@ DEF_OP(Vector_FToF) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
@@ -186,31 +223,31 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
|
||||
+1
-1
@@ -146,7 +146,7 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
|
||||
@@ -154,8 +154,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
@@ -181,13 +179,10 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(SPLATVECTOR2, SplatVector);
|
||||
REGISTER_OP(SPLATVECTOR4, SplatVector);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
@@ -246,18 +241,13 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VINSSCALARELEMENT, VInsScalarElement);
|
||||
REGISTER_OP(VEXTRACTELEMENT, VExtractElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VSLI, VSLI);
|
||||
REGISTER_OP(VSRI, VSRI);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VBITCAST, VBitcast);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
|
||||
@@ -49,7 +49,7 @@ namespace FEXCore::CPU {
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
@@ -181,8 +181,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -207,7 +205,6 @@ namespace FEXCore::CPU {
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -264,18 +261,13 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
@@ -25,93 +25,139 @@ static inline void CacheLineFlush(char *Addr) {
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (IROp->Size) {
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
@@ -144,8 +190,8 @@ DEF_OP(StoreFlag) {
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -158,7 +204,8 @@ DEF_OP(LoadMem) {
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
memset(GDP, 0, 16);
|
||||
|
||||
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
|
||||
@@ -180,16 +227,15 @@ DEF_OP(LoadMem) {
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -221,41 +267,11 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
STORE_DATA(1, uint8_t)
|
||||
STORE_DATA(2, uint16_t)
|
||||
STORE_DATA(4, uint32_t)
|
||||
STORE_DATA(8, uint64_t)
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
|
||||
}
|
||||
#undef STORE_DATA
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
|
||||
@@ -19,13 +19,6 @@ $end_info$
|
||||
#include <sys/random.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(Fence) {
|
||||
|
||||
@@ -30,13 +30,6 @@ DEF_OP(CreateElementPair) {
|
||||
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+851
-598
File diff suppressed because it is too large.
Load diff
+73
-19
@@ -14,7 +14,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
@@ -1168,25 +1168,79 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V16B(), Op->Index);
|
||||
break;
|
||||
case 2:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V8H(), Op->Index);
|
||||
break;
|
||||
case 4:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V4S(), Op->Index);
|
||||
break;
|
||||
case 8:
|
||||
umov(GetReg<RA_64>(Node), GetSrc(Op->Vector.ID()).V2D(), Op->Index);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
|
||||
break;
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = Op->Header.ElementSize * 8;
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
const auto PerformMove = [&](const aarch64::VRegister& reg, int index) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov(GetReg<RA_32>(Node), reg.V16B(), index);
|
||||
break;
|
||||
case 2:
|
||||
umov(GetReg<RA_32>(Node), reg.V8H(), index);
|
||||
break;
|
||||
case 4:
|
||||
umov(GetReg<RA_32>(Node), reg.V4S(), index);
|
||||
break;
|
||||
case 8:
|
||||
umov(GetReg<RA_64>(Node), reg.V2D(), index);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
if (Offset < SSERegBitSize) {
|
||||
// Desired data lies within the lower 128-bit lane, so we
|
||||
// can treat the operation as a 128-bit operation, even
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE,
|
||||
"Host doesn't support SVE. Cannot perform 256-bit operation.");
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit,
|
||||
"Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize,
|
||||
"Trying to extract element outside bounds of register. Offset={}, Index={}",
|
||||
Offset, Op->Index);
|
||||
|
||||
// We need to use the upper 128-bit lane, so lets move it down.
|
||||
// Inverting our dedicated predicate for 128-bit operations selects
|
||||
// all of the top lanes. We can then compact those into a temporary.
|
||||
const auto CompactPred = p0;
|
||||
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
|
||||
compact(VTMP1.Z().VnD(), CompactPred, Vector.Z().VnD());
|
||||
|
||||
// Sanitize the zero-based index to work on the now-moved
|
||||
// upper half of the vector.
|
||||
const auto SanitizedIndex = [OpSize, Op] {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
return Op->Index - 16;
|
||||
case 2:
|
||||
return Op->Index - 8;
|
||||
case 4:
|
||||
return Op->Index - 4;
|
||||
case 8:
|
||||
return Op->Index - 2;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled OpSize: {}", OpSize);
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
// Move the value from the now-low-lane data.
|
||||
PerformMove(VTMP1, SanitizedIndex);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
// Size is the size of each pair element
|
||||
|
||||
@@ -19,7 +19,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -40,7 +40,7 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
@@ -197,7 +197,11 @@ DEF_OP(Syscall) {
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
|
||||
add(sp, sp, SPOffset);
|
||||
|
||||
@@ -239,7 +243,6 @@ DEF_OP(InlineSyscall) {
|
||||
bool Intersects{};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
@@ -381,7 +384,11 @@ DEF_OP(Thunk) {
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -448,7 +455,11 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
@@ -459,6 +470,7 @@ DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
@@ -467,10 +479,13 @@ DEF_OP(CPUID) {
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0, x1
|
||||
|
||||
+460
-101
@@ -10,28 +10,117 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
// This is going to be a little gross. Pls forgive me.
|
||||
// Since SVE has the whole vector length agnostic programming
|
||||
// thing going on, we can't exactly freely insert entries into
|
||||
// arbitrary locations in the vector.
|
||||
//
|
||||
// SVE *does* have INSR, however this only shifts the entire
|
||||
// vector to the left by an element size and inserts a value
|
||||
// at the beginning of the vector. Not *quite* what we need.
|
||||
// (though INSR *is* very useful for other things).
|
||||
//
|
||||
// The idea is (in the case of the upper lane), move the upper
|
||||
// lane down, insert into it and recombine with the lower lane.
|
||||
//
|
||||
// In the case of the lower lane, insert and then recombine with
|
||||
// the upper lane.
|
||||
|
||||
if (InUpperLane) {
|
||||
// Move the upper lane down for the insertion.
|
||||
const auto CompactPred = p0;
|
||||
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
|
||||
compact(VTMP1.Z().VnD(), CompactPred, DestVector.Z().VnD());
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
|
||||
// Put data in place for destructive SPLICE below.
|
||||
mov(Dst.Z().VnD(), DestVector.Z().VnD());
|
||||
|
||||
// Inserts the GPR value into the given V register.
|
||||
// Also automatically adjusts the index in the case of using the
|
||||
// moved upper lane.
|
||||
const auto Insert = [&](const aarch64::VRegister& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
ins(reg.V16B(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 2:
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
ins(reg.V8H(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 4:
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
ins(reg.V4S(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 8:
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
ins(reg.V2D(), index, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
Insert(VTMP1, DestIdx);
|
||||
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), VTMP1.Z().VnD());
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), DestVector.Z().VnD());
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
} else {
|
||||
mov(Dst, DestVector);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
ins(Dst.V16B(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(Dst.V8H(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(Dst.V4S(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(Dst.V2D(), DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,8 +146,11 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
@@ -76,6 +168,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -96,116 +192,379 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
scvtf(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
scvtf(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
scvtf(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
scvtf(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
fcvtzs(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
fcvtzs(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
fcvtzs(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
fcvtzs(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
fcvtzs(Dst.Z().VnH(), Mask, Dst.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
fcvtzs(Dst.Z().VnS(), Mask, Dst.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
fcvtzs(Dst.Z().VnD(), Mask, Dst.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.V8H(), Vector.V8H());
|
||||
fcvtzs(Dst.V8H(), Dst.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.V4S(), Vector.V4S());
|
||||
fcvtzs(Dst.V4S(), Dst.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.V2D(), Vector.V2D());
|
||||
fcvtzs(Dst.V2D(), Dst.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// Curiously, FCVTLT and FCVTNT have no bottom variants,
|
||||
// and also interesting is that FCVTLT will iterate the
|
||||
// source vector by accessing each odd element and storing
|
||||
// them consecutively in the destination.
|
||||
//
|
||||
// FCVTNT is somewhat like the opposite. It will read each
|
||||
// consecutive element, but store each result into every odd
|
||||
// element in the destination vector.
|
||||
//
|
||||
// We need to undo the behavior of FCVTNT with UZP2. In the case
|
||||
// of FCVTLT, we instead need to set the vector up with ZIP1, so
|
||||
// that the elements will be processed correctly.
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(Dst.Z().VnH(), Vector.Z().VnH(), Vector.Z().VnH());
|
||||
fcvtlt(Dst.Z().VnS(), Mask, Dst.Z().VnH());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(Dst.Z().VnS(), Vector.Z().VnS(), Vector.Z().VnS());
|
||||
fcvtlt(Dst.Z().VnD(), Mask, Dst.Z().VnS());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(Dst.Z().VnH(), Mask, Vector.Z().VnS());
|
||||
uzp2(Dst.Z().VnH(), Dst.Z().VnH(), Dst.Z().VnH());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(Dst.Z().VnS(), Mask, Vector.Z().VnD());
|
||||
uzp2(Dst.Z().VnS(), Dst.Z().VnS(), Dst.Z().VnS());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
} else {
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
fcvtl(Dst.V4S(), Vector.V4H());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(Dst.V2D(), Vector.V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtn(Dst.V4H(), Vector.V4S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(Dst.V2S(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintn(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintn(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintn(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintm(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintm(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintm(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintp(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintp(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintp(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintz(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintz(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintz(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
}
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintn(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintn(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintm(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintm(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintp(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintp(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintz(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintz(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
@@ -18,7 +18,7 @@ DEF_OP(AESImc) {
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -27,7 +27,7 @@ DEF_OP(AESEnc) {
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -35,7 +35,7 @@ DEF_OP(AESEncLast) {
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -44,7 +44,7 @@ DEF_OP(AESDec) {
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -52,7 +52,7 @@ DEF_OP(AESDecLast) {
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label PastConstant;
|
||||
@@ -69,9 +69,8 @@ DEF_OP(AESKeyGenAssist) {
|
||||
if (Op->RCON) {
|
||||
tbl(VTMP1.V16B(), VTMP1.V16B(), VTMP3.V16B());
|
||||
|
||||
LoadConstant(TMP1.W(), Op->RCON);
|
||||
ins(VTMP2.V4S(), 1, TMP1.W());
|
||||
ins(VTMP2.V4S(), 3, TMP1.W());
|
||||
LoadConstant(TMP1, static_cast<uint64_t>(Op->RCON) << 32);
|
||||
dup(VTMP2.V2D(), TMP1);
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), VTMP2.V16B());
|
||||
}
|
||||
else {
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
|
||||
+79
-13
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -79,7 +80,7 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -94,7 +95,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, uint16_t>(x1);
|
||||
#else
|
||||
blr(x1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -109,7 +115,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, float>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -128,7 +138,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, double>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -153,7 +167,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint32_t>(x1);
|
||||
#else
|
||||
blr(x1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -174,7 +192,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<float, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -193,7 +215,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -210,7 +236,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -229,7 +259,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double, double>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -249,7 +283,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -267,7 +305,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -285,7 +327,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -306,8 +352,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t, uint64_t>(x4);
|
||||
#else
|
||||
blr(x4);
|
||||
|
||||
#endif
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
@@ -324,7 +373,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -347,7 +400,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t, uint64_t, uint64_t>(x4);
|
||||
#else
|
||||
blr(x4);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -420,7 +477,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
@@ -473,7 +530,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -494,7 +551,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
@@ -511,6 +568,7 @@ void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
@@ -519,6 +577,7 @@ void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -530,7 +589,7 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
@@ -673,6 +732,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
@@ -725,10 +786,12 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
sub(sp, sp, SpillSlots * 16);
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
sub(sp, sp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
LoadConstant(x0, TotalSpillSlotsSize);
|
||||
sub(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
@@ -796,14 +859,17 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
if (SpillSlots == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
add(sp, sp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
LoadConstant(x0, TotalSpillSlotsSize);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
+13
-22
@@ -23,16 +23,6 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
#define TMP3 x2
|
||||
#define TMP4 x3
|
||||
|
||||
#define VTMP1 v1
|
||||
#define VTMP2 v2
|
||||
#define VTMP3 v3
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
@@ -128,6 +118,17 @@ private:
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]] SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
|
||||
aarch64::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
@@ -212,7 +213,7 @@ private:
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
@@ -224,7 +225,7 @@ private:
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
@@ -344,8 +345,6 @@ private:
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -365,13 +364,10 @@ private:
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector2);
|
||||
DEF_OP(SplatVector4);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -430,18 +426,13 @@ private:
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
+758
-365
File diff suppressed because it is too large.
Load diff
+21
-14
@@ -11,7 +11,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
@@ -41,16 +41,19 @@ DEF_OP(Break) {
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Op->Reason.Signal,
|
||||
.TrapNo = Op->Reason.TrapNumber,
|
||||
.si_code = Op->Reason.si_code,
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(x1, Constant);
|
||||
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData)));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
@@ -119,8 +122,11 @@ DEF_OP(SetRoundingMode) {
|
||||
|
||||
mrs(TMP1, FPCR);
|
||||
|
||||
// vixl simulator doesn't support anything beyond ties-to-even rounding
|
||||
#ifndef VIXL_SIMULATOR
|
||||
// Insert the rounding flags
|
||||
bfi(TMP1, TMP2, 22, 2);
|
||||
#endif
|
||||
|
||||
// Insert the FTZ flag
|
||||
lsr(TMP2, Src, 2);
|
||||
@@ -134,6 +140,7 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
@@ -145,10 +152,10 @@ DEF_OP(Print) {
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
blr(x3);
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
@@ -68,17 +68,11 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+4379
-1627
File diff suppressed because it is too large.
Load diff
+44
-11
@@ -1143,26 +1143,59 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
|
||||
const auto Is256Bit = Offset >= SSEBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrb(GetDst<RA_32>(Node), xmm15, Op->Index - 16);
|
||||
} else {
|
||||
pextrb(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrw(GetDst<RA_32>(Node), xmm15, Op->Index - 8);
|
||||
} else {
|
||||
pextrw(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrd(GetDst<RA_32>(Node), xmm15, Op->Index - 4);
|
||||
} else {
|
||||
pextrd(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrq(GetDst<RA_64>(Node), xmm15, Op->Index - 2);
|
||||
} else {
|
||||
pextrq(GetDst<RA_64>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
@@ -42,7 +42,7 @@ DEF_OP(SignalReturn) {
|
||||
DEF_OP(CallbackReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
@@ -71,7 +71,7 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
uint64_t NewRIP;
|
||||
|
||||
+224
-74
@@ -17,27 +17,76 @@ namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
if (InUpperLane && !Is256Bit) {
|
||||
LOGMAN_MSG_A_FMT("Attempt to access upper 128-bit lane in 128-bit operation! Offset={}",
|
||||
Offset);
|
||||
return;
|
||||
}
|
||||
|
||||
if (Is256Bit) {
|
||||
vmovapd(ToYMM(Dst), ToYMM(DestVector));
|
||||
} else {
|
||||
vmovapd(Dst, DestVector);
|
||||
}
|
||||
|
||||
const auto Insert = [&](const Xbyak::Xmm& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
pinsrb(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
pinsrw(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
pinsrd(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
pinsrq(reg, GetSrc<RA_64>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
vextracti128(xmm15, ToYMM(Dst), 1);
|
||||
Insert(xmm15, DestIdx);
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,8 +112,10 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
@@ -83,6 +134,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -104,99 +159,194 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtdq2ps(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtdq2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
// There is no vector form of this instruction until AVX512VL + AVX512DQ (vcvtqq2pd)
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
pextrq(rax, Vector, 1);
|
||||
pextrq(rcx, Vector, 0);
|
||||
cvtsi2sd(Dst, rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
vmovlhps(Dst, Dst, xmm15);
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
|
||||
pextrq(rax, xmm15, 1);
|
||||
pextrq(rcx, xmm15, 0);
|
||||
cvtsi2sd(xmm15, rcx);
|
||||
cvtsi2sd(xmm14, rax);
|
||||
movlhps(xmm15, xmm14);
|
||||
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvttps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvttpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvtpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtps2pd(ToYMM(Dst), Vector);
|
||||
} else {
|
||||
vcvtps2pd(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtpd2ps(Dst, ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t RoundMode{};
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
RoundMode = 0b0000'0'0'00;
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'01;
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'10;
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
RoundMode = 0b0000'0'0'11;
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
RoundMode = 0b0000'0'1'00;
|
||||
break;
|
||||
}
|
||||
const uint8_t RoundMode = [Op] {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
return 0b0000'0'0'00;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
return 0b0000'0'0'01;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
return 0b0000'0'0'10;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
return 0b0000'0'0'11;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
return 0b0000'0'1'00;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled rounding mode");
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundps(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundps(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundpd(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundpd(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+30
-17
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -60,32 +61,42 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
const auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
sub(rsp, AVXRegSize * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
movaps(ptr[rsp + i * 16], RAXMM_x[i]);
|
||||
vmovups(ptr[rsp + i * AVXRegSize], ToYMM(RAXMM_x[i]));
|
||||
}
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
for (const auto &Reg : RA64) {
|
||||
push(Reg);
|
||||
}
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
const auto NumPush = RA64.size();
|
||||
if ((NumPush & 1) != 0) {
|
||||
// Align
|
||||
sub(rsp, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void X86JITCore::PopRegs() {
|
||||
auto NumPush = RA64.size();
|
||||
const auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
movaps(RAXMM_x[i], ptr[rsp + i * 16]);
|
||||
if ((NumPush & 1) != 0) {
|
||||
// Align
|
||||
add(rsp, 8);
|
||||
}
|
||||
|
||||
add(rsp, 16 * RAXMM_x.size());
|
||||
for (uint32_t i = RA64.size(); i > 0; --i) {
|
||||
pop(RA64[i - 1]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
vmovups(ToYMM(RAXMM_x[i]), ptr[rsp + i * AVXRegSize]);
|
||||
}
|
||||
|
||||
add(rsp, AVXRegSize * RAXMM_x.size());
|
||||
}
|
||||
|
||||
void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
@@ -360,7 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
|
||||
{
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -572,6 +583,8 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("x86::CompileCode");
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
@@ -599,7 +612,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
sub(rsp, SpillSlots * 16);
|
||||
sub(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
|
||||
@@ -209,7 +209,7 @@ private:
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
using JCC = void (X86JITCore::*)(const Label& label, LabelType type);
|
||||
@@ -337,6 +337,8 @@ private:
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
@@ -345,8 +347,6 @@ private:
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -366,12 +366,10 @@ private:
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -430,18 +428,13 @@ private:
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
+365
-189
@@ -21,130 +21,266 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(GetDst<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(GetDst<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(GetDst<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid GPR load of size 16");
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(GetDst(Node), dword [STATE + Op->Offset]);
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(GetDst(Node), qword [STATE + Op->Offset]);
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16:
|
||||
LogMan::Msg::DFmt("Invalid store size of 16");
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid store size of 16");
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrb(byte [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrw(word [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovd(dword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovq(qword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetSrc<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
movzx(GetSrc<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
mov(GetSrc<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
mov(GetSrc<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetSrc(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -153,21 +289,21 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 2:
|
||||
movzx(GetDst<RA_32>(Node), word [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), word [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 4:
|
||||
mov(GetDst<RA_32>(Node), dword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_32>(Node), dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_64>(Node), qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -186,53 +322,63 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(eax, byte [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, byte [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(eax, word [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, word [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [rax + index * Op->Stride]);
|
||||
vmovd(Dst, dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
|
||||
vmovq(Dst, qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pinsrb(GetDst(Node), byte [STATE + rax], 0);
|
||||
pinsrb(Dst, byte [STATE + rax], 0);
|
||||
break;
|
||||
case 2:
|
||||
pinsrw(GetDst(Node), word [STATE + rax], 0);
|
||||
pinsrw(Dst, word [STATE + rax], 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [STATE + rax]);
|
||||
vmovd(Dst, dword [STATE + rax]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [STATE + rax]);
|
||||
vmovq(Dst, qword [STATE + rax]);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + rax]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + rax]);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + rax]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + rax]);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
vmovups(ToYMM(Dst), yword [STATE + rax]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -245,12 +391,13 @@ DEF_OP(LoadContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
size_t size = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto value = GetSrc<RA_64>(Op->Value.ID());
|
||||
const auto Value = GetSrc<RA_64>(Op->Value.ID());
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -258,10 +405,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
if (!(OpSize == 1 || OpSize == 2 || OpSize == 4 || OpSize == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
}
|
||||
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
mov(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -270,57 +417,64 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(xword [STATE + rax], value);
|
||||
else
|
||||
movups(xword [STATE + rax], value);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(xword [STATE + rax], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + rax], Value);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [STATE + rax], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -333,10 +487,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
@@ -355,36 +509,44 @@ DEF_OP(SpillRegister) {
|
||||
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Src = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(dword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movss(dword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(qword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movsd(qword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(xword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movaps(xword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(yword [rsp + SlotOffset], ToYMM(Src));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
DEF_OP(FillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
@@ -403,23 +565,33 @@ DEF_OP(FillRegister) {
|
||||
mov(GetDst<RA_64>(Node), qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(GetDst(Node), dword [rsp + SlotOffset]);
|
||||
vmovss(Dst, dword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(GetDst(Node), qword [rsp + SlotOffset]);
|
||||
vmovsd(Dst, qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(GetDst(Node), xword [rsp + SlotOffset]);
|
||||
vmovaps(Dst, xword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
@@ -464,130 +636,136 @@ Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
const auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx (Dst, byte [MemPtr]);
|
||||
movzx(Dst, byte [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx (Dst, word [MemPtr]);
|
||||
movzx(Dst, word [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(Dst.cvt32(), dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
auto Dst = GetDst(Node);
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(eax, byte [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(eax, word [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(Dst, dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
else
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, GetDst(Node));
|
||||
}
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
vmovups(Dst, xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 2:
|
||||
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 4:
|
||||
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 8:
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrb(byte [MemPtr], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrw(word [MemPtr], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovd(dword [MemPtr], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovq(qword [MemPtr], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
vmovups(xword [MemPtr], Value);
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [MemPtr], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
@@ -618,8 +796,8 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP(LOADREGISTER, Unhandled); // SRA specific, not supported on this backend
|
||||
REGISTER_OP(STOREREGISTER, Unhandled);
|
||||
REGISTER_OP(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
@@ -630,8 +808,6 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
|
||||
@@ -47,14 +47,22 @@ DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)], Op->Reason.Signal);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], Op->Reason.TrapNumber);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], Op->Reason.ErrorRegister);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], Op->Reason.si_code);
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Op->Reason.Signal,
|
||||
.TrapNo = Op->Reason.TrapNumber,
|
||||
.si_code = Op->Reason.si_code,
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
mov(TMP1, Constant);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData)], TMP1);
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
|
||||
@@ -73,17 +73,11 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+2975
-1095
File diff suppressed because it is too large.
Load diff
+9
-8
@@ -8,6 +8,7 @@
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -17,7 +18,7 @@ namespace Context {
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
struct LookupCacheEntry {
|
||||
struct LookupCacheEntry {
|
||||
uintptr_t HostCode;
|
||||
uintptr_t GuestCode;
|
||||
};
|
||||
@@ -54,7 +55,7 @@ public:
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
@@ -62,7 +63,7 @@ public:
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
@@ -73,7 +74,7 @@ public:
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -88,7 +89,7 @@ public:
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
@@ -158,7 +159,7 @@ public:
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
@@ -167,7 +168,7 @@ public:
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
@@ -239,7 +240,7 @@ private:
|
||||
|
||||
|
||||
std::map<BlockLinkTag, std::function<void()>> BlockLinks;
|
||||
std::map<uint64_t, uint64_t> BlockList;
|
||||
tsl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
|
||||
+857
-364
File diff suppressed because it is too large.
Load diff
+37
-3
@@ -76,10 +76,10 @@ public:
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::Context *CTX{};
|
||||
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
@@ -278,6 +278,12 @@ public:
|
||||
void NOTOp(OpcodeArgs);
|
||||
void XADDOp(OpcodeArgs);
|
||||
void PopcountOp(OpcodeArgs);
|
||||
void DAAOp(OpcodeArgs);
|
||||
void DASOp(OpcodeArgs);
|
||||
void AAAOp(OpcodeArgs);
|
||||
void AASOp(OpcodeArgs);
|
||||
void AAMOp(OpcodeArgs);
|
||||
void AADOp(OpcodeArgs);
|
||||
void XLATOp(OpcodeArgs);
|
||||
template<bool Reseed>
|
||||
void RDRANDOp(OpcodeArgs);
|
||||
@@ -292,6 +298,8 @@ public:
|
||||
void WriteSegmentReg(OpcodeArgs);
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
|
||||
// SSE
|
||||
void MOVAPSOp(OpcodeArgs);
|
||||
void MOVUPSOp(OpcodeArgs);
|
||||
@@ -398,6 +406,26 @@ public:
|
||||
// ADX Ops
|
||||
void ADXOp(OpcodeArgs);
|
||||
|
||||
// AVX Ops
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
|
||||
void VANDNOp(OpcodeArgs);
|
||||
|
||||
void VMOVAPS_VMOVAPD_Op(OpcodeArgs);
|
||||
void VMOVUPS_VMOVUPD_Op(OpcodeArgs);
|
||||
|
||||
void VMOVHPOp(OpcodeArgs);
|
||||
void VMOVLPOp(OpcodeArgs);
|
||||
|
||||
void VMOVDDUPOp(OpcodeArgs);
|
||||
void VMOVSHDUPOp(OpcodeArgs);
|
||||
void VMOVSLDUPOp(OpcodeArgs);
|
||||
|
||||
void VMOVVectorNTOp(OpcodeArgs);
|
||||
|
||||
void VZEROOp(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
template<size_t width>
|
||||
void FLD(OpcodeArgs);
|
||||
@@ -515,7 +543,7 @@ public:
|
||||
void X87FRSTORF64(OpcodeArgs);
|
||||
void X87FXAMF64(OpcodeArgs);
|
||||
void X87LDENVF64(OpcodeArgs);
|
||||
|
||||
|
||||
template<size_t width, bool Integer, FCOMIFlags whichflags, bool poptwice>
|
||||
void FCOMIF64(OpcodeArgs);
|
||||
|
||||
@@ -646,6 +674,7 @@ private:
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
@@ -657,6 +686,11 @@ private:
|
||||
// Non-temporal streaming
|
||||
ACCESS_STREAM,
|
||||
};
|
||||
OrderedNode *LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0);
|
||||
OrderedNode *LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, OrderedNode *const Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, OrderedNode *const Src);
|
||||
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
@@ -221,7 +221,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *XMM0 = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
|
||||
@@ -769,7 +769,11 @@ void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, Orde
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
|
||||
auto OpSize = SrcSize * 8;
|
||||
if (OpSize < Shift) {
|
||||
Shift &= (OpSize - 1);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, OpSize - Shift, Src1));
|
||||
}
|
||||
|
||||
// PF
|
||||
@@ -934,6 +938,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
|
||||
// If shift == 0, don't update flags
|
||||
@@ -963,7 +968,9 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode
|
||||
// OF
|
||||
{
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF);
|
||||
@@ -977,8 +984,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
|
||||
auto NewCF = _Bfe(1, OpSize - 1, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -989,8 +995,10 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Res), NewCF));
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1000,17 +1008,22 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, Ord
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, 0, Res);
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
}
|
||||
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is the top two MSBs XOR'd together
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1)));
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+243
-77
@@ -33,11 +33,46 @@ void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVVectorNTOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1, true, false, MemoryAccessType::ACCESS_STREAM);
|
||||
|
||||
// TODO: When stores and loads gain the ability to explicitly express
|
||||
// whether a vector extension or an insert is desirable, ensure
|
||||
// the 128-bit case here is a zero extend on store if the destination
|
||||
// is a register.
|
||||
|
||||
StoreResult(FPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVAPSOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVAPS_VMOVAPD_Op(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto Is128BitDest = GetDstSize(Op) == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
if (Op->Dest.IsGPR() && Is128BitDest) {
|
||||
// Perform 32 byte store to clear the upper lane.
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 32, -1);
|
||||
} else {
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVUPS_VMOVUPD_Op(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1);
|
||||
const auto Is128BitDest = GetDstSize(Op) == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
if (Op->Dest.IsGPR() && Is128BitDest) {
|
||||
// Perform 32 byte store to clear the upper lane.
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 32, 1);
|
||||
} else {
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVUPSOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1);
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
@@ -68,20 +103,33 @@ void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVHPOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 16);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 8, 1, 0, Src1, Src2);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
} else {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 16);
|
||||
OrderedNode *Result = _VInsElement(16, 8, 0, 1, Src, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
if (Op->Dest.IsGPR()) {
|
||||
// xmm, xmm is movhlps special case
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16);
|
||||
Src = _VExtractElement(16, 8, Src, 1);
|
||||
auto Result = _VInsScalarElement(16, 8, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 8, 0, 1, Dest, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 16, 16);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16);
|
||||
auto Result = _VInsScalarElement(16, 8, 0, Dest, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 16);
|
||||
auto DstSize = GetDstSize(Op);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
auto Result = _VInsElement(16, 8, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -89,30 +137,68 @@ void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVLPOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 16);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 8, 0, 0, Src1, Src2);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
} else {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 8, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSHDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 4, 3, 3, Src, Src);
|
||||
Result = _VInsElement(16, 4, 2, 3, Result, Src);
|
||||
Result = _VInsElement(16, 4, 1, 1, Result, Src);
|
||||
OrderedNode *Result = _VInsElement(16, 4, 2, 3, Src, Src);
|
||||
Result = _VInsElement(16, 4, 0, 1, Result, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVSHDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
OrderedNode *Result = _VInsElement(SrcSize, 4, 2, 3, Src, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 0, 1, Result, Src);
|
||||
if (Is256Bit) {
|
||||
Result = _VInsElement(SrcSize, 4, 4, 5, Result, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 6, 7, Result, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSLDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 4, 3, 2, Src, Src);
|
||||
Result = _VInsElement(16, 4, 2, 2, Result, Src);
|
||||
Result = _VInsElement(16, 4, 1, 0, Result, Src);
|
||||
Result = _VInsElement(16, 4, 0, 0, Result, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVSLDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
OrderedNode *Result = _VInsElement(SrcSize, 4, 3, 2, Src, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 1, 0, Result, Src);
|
||||
if (Is256Bit) {
|
||||
Result = _VInsElement(SrcSize, 4, 5, 4, Result, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 7, 6, Result, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) {
|
||||
// MOVSS xmm1, xmm2
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
auto Result = _VInsScalarElement(16, 4, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 4, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else if (Op->Dest.IsGPR()) {
|
||||
@@ -133,7 +219,7 @@ void OpDispatchBuilder::MOVSDOp(OpcodeArgs) {
|
||||
// xmm1[63:0] <- xmm2[63:0]
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Result = _VInsScalarElement(16, 8, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 8, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else if (Op->Dest.IsGPR()) {
|
||||
@@ -302,6 +388,48 @@ void OpDispatchBuilder::VectorALUOp<IR::OP_VUQSUB, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUOp<IR::OP_VUQSUB, 2>(OpcodeArgs);
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorALUOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _VAdd(Size, ElementSize, Src1, Src2);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
if (Is128Bit) {
|
||||
// 128-bit variants need to zero the upper lane.
|
||||
Result = _VMov(Size, ALUOp);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VAND, 16>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VOR, 16>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VXOR, 16>(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorALUROp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
@@ -322,8 +450,9 @@ void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 8>(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -333,9 +462,9 @@ void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
|
||||
if (Size != ElementSize) {
|
||||
if (DstSize != ElementSize) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsScalarElement(Size, ElementSize, 0, Dest, Result);
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -369,11 +498,12 @@ void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 8>(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
auto DstSize = GetDstSize(Op);
|
||||
if constexpr (Scalar) {
|
||||
Size = ElementSize;
|
||||
}
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _VFSqrt(Size, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
@@ -381,7 +511,7 @@ void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
auto Result = _VInsScalarElement(GetSrcSize(Op), ElementSize, 0, Dest, ALUOp);
|
||||
auto Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else {
|
||||
@@ -442,14 +572,8 @@ void OpDispatchBuilder::MOVQOp(OpcodeArgs) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
const auto fprLowOffset = CTX->HostFeatures.SupportsAVX ? offsetof(Core::CPUState, xmm.avx.data[gprIndex][0])
|
||||
: offsetof(Core::CPUState, xmm.sse.data[gprIndex][0]);
|
||||
const auto fprHighOffset = CTX->HostFeatures.SupportsAVX ? offsetof(Core::CPUState, xmm.avx.data[gprIndex][1])
|
||||
: offsetof(Core::CPUState, xmm.sse.data[gprIndex][1]);
|
||||
|
||||
_StoreContext(8, FPRClass, Src, fprLowOffset);
|
||||
auto Const = _Constant(0);
|
||||
_StoreContext(8, GPRClass, Const, fprHighOffset);
|
||||
auto Reg = _VMov(16, Src);
|
||||
StoreXMMRegister(gprIndex, Reg);
|
||||
}
|
||||
else {
|
||||
// This is simple, just store the result
|
||||
@@ -660,6 +784,22 @@ void OpDispatchBuilder::ANDNOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VANDNOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
|
||||
Src1 = _VNot(Size, Size, Src1);
|
||||
OrderedNode *Dest = _VAnd(Size, Size, Src1, Src2);
|
||||
if (Is128Bit) {
|
||||
Dest = _VMov(16, Dest);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PINSROp(OpcodeArgs) {
|
||||
auto Size = GetDstSize(Op);
|
||||
@@ -927,7 +1067,11 @@ void OpDispatchBuilder::PSRLDQ(OpcodeArgs) {
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
|
||||
auto Result = _VSRI(Size, 16, Dest, Shift);
|
||||
OrderedNode *Result = _VectorZero(Size);
|
||||
if (Shift < Size) {
|
||||
Result = _VExtr(Size, 1, Result, Dest, Shift);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -939,7 +1083,10 @@ void OpDispatchBuilder::PSLLDQ(OpcodeArgs) {
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
|
||||
auto Result = _VSLI(Size, 16, Dest, Shift);
|
||||
OrderedNode *Result = _VectorZero(Size);
|
||||
if (Shift < Size) {
|
||||
Result = _VExtr(Size, 1, Dest, Result, Size - Shift);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -978,10 +1125,28 @@ void OpDispatchBuilder::PAVGOp<2>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Res = _SplatVector2(Src);
|
||||
OrderedNode *Res = _VDupElement(16, GetSrcSize(Op), Src, 0);
|
||||
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVDDUPOp(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto IsSrcGPR = Op->Src[0].IsGPR();
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto MemSize = Is256Bit ? 32 : 8;
|
||||
|
||||
OrderedNode *Src = IsSrcGPR ? LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1)
|
||||
: LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], MemSize, Op->Flags, -1);
|
||||
|
||||
OrderedNode *Res = _VInsElement(SrcSize, 8, 1, 0, Src, Src);
|
||||
if (Is256Bit) {
|
||||
Res = _VInsElement(SrcSize, 8, 3, 2, Res, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Res, 32, -1);
|
||||
}
|
||||
|
||||
template<size_t DstElementSize>
|
||||
void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
@@ -992,7 +1157,7 @@ void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
|
||||
Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src);
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
@@ -1087,13 +1252,16 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
Src = _Float_FToF(DstElementSize, SrcElementSize, Src);
|
||||
Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src);
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
auto Result = _VInsElement(DstSize, DstElementSize, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
@@ -1103,8 +1271,9 @@ void OpDispatchBuilder::Scalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
size_t Size = GetDstSize(Op);
|
||||
|
||||
if constexpr (DstElementSize > SrcElementSize) {
|
||||
Src = _Vector_FToF(Size, SrcElementSize << 1, Src, SrcElementSize);
|
||||
@@ -1181,13 +1350,12 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
// Until we get correct PHI nodes this is required to be a loop unroll
|
||||
const auto GPRSize = CTX->GetGPRSize();
|
||||
const auto Size = uint32_t{GetSrcSize(Op)} * 8;
|
||||
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
OrderedNode *MemDest = _LoadContext(GPRSize, GPRClass, GPROffset(X86State::REG_RDI));
|
||||
auto MemDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
const size_t NumElements = Size / 64;
|
||||
for (size_t Element = 0; Element < NumElements; ++Element) {
|
||||
@@ -1281,7 +1449,7 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Result);
|
||||
Result = _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Dest, Result);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -1398,16 +1566,9 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
const auto GetXMMOffset = [this](size_t i) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
return offsetof(Core::CPUState, xmm.avx.data[i]);
|
||||
} else {
|
||||
return offsetof(Core::CPUState, xmm.sse.data[i]);
|
||||
}
|
||||
};
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = _LoadContext(16, FPRClass, GetXMMOffset(i));
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
@@ -1455,18 +1616,11 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
const auto GetXMMOffset = [this](size_t i) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
return offsetof(Core::CPUState, xmm.avx.data[i]);
|
||||
} else {
|
||||
return offsetof(Core::CPUState, xmm.sse.data[i]);
|
||||
}
|
||||
};
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, XMMReg, GetXMMOffset(i));
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1563,6 +1717,9 @@ void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
|
||||
static_assert(ElementSize == sizeof(uint32_t),
|
||||
"Currently only handles 32-bit -> 64-bit");
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
@@ -1579,17 +1736,8 @@ void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
OrderedNode* Srcs1[2]{};
|
||||
OrderedNode* Srcs2[2]{};
|
||||
|
||||
Srcs1[0] = _VExtr(Size, ElementSize, Src1, Src1, 0);
|
||||
Srcs1[1] = _VExtr(Size, ElementSize, Src1, Src1, 2);
|
||||
|
||||
Srcs2[0] = _VExtr(Size, ElementSize, Src2, Src2, 0);
|
||||
Srcs2[1] = _VExtr(Size, ElementSize, Src2, Src2, 2);
|
||||
|
||||
Src1 = _VInsElement(Size, ElementSize, 1, 0, Srcs1[0], Srcs1[1]);
|
||||
Src2 = _VInsElement(Size, ElementSize, 1, 0, Srcs2[0], Srcs2[1]);
|
||||
Src1 = _VInsElement(Size, ElementSize, 1, 2, Src1, Src1);
|
||||
Src2 = _VInsElement(Size, ElementSize, 1, 2, Src2, Src2);
|
||||
|
||||
if constexpr (Signed) {
|
||||
Res = _VSMull(Size, ElementSize, Src1, Src2);
|
||||
@@ -1613,11 +1761,9 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) {
|
||||
// This instruction is a bit special in that if the source is MMX then it zexts to 128bit
|
||||
if constexpr (ToXMM) {
|
||||
const auto Index = Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0;
|
||||
const auto Offset = CTX->HostFeatures.SupportsAVX ? offsetof(FEXCore::Core::CPUState, xmm.avx.data[Index][0])
|
||||
: offsetof(FEXCore::Core::CPUState, xmm.sse.data[Index][0]);
|
||||
|
||||
Src = _VMov(16, Src);
|
||||
_StoreContext(16, FPRClass, Src, Offset);
|
||||
StoreXMMRegister(Index, Src);
|
||||
}
|
||||
else {
|
||||
// This is simple, just store the result
|
||||
@@ -1714,8 +1860,8 @@ void OpDispatchBuilder::PFNACCOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode *ResSubSrc{};
|
||||
OrderedNode *ResSubDest{};
|
||||
auto UpperSubDest = _VExtractElement(Size, 4, Dest, 1);
|
||||
auto UpperSubSrc = _VExtractElement(Size, 4, Src, 1);
|
||||
auto UpperSubDest = _VDupElement(Size, 4, Dest, 1);
|
||||
auto UpperSubSrc = _VDupElement(Size, 4, Src, 1);
|
||||
|
||||
ResSubDest = _VFSub(4, 4, Dest, UpperSubDest);
|
||||
ResSubSrc = _VFSub(4, 4, Src, UpperSubSrc);
|
||||
@@ -1733,7 +1879,7 @@ void OpDispatchBuilder::PFPNACCOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode *ResAdd{};
|
||||
OrderedNode *ResSub{};
|
||||
auto UpperSubDest = _VExtractElement(Size, 4, Dest, 1);
|
||||
auto UpperSubDest = _VDupElement(Size, 4, Dest, 1);
|
||||
|
||||
ResSub = _VFSub(4, 4, Dest, UpperSubDest);
|
||||
ResAdd = _VFAddP(Size, 4, Src, Src);
|
||||
@@ -1866,8 +2012,6 @@ void OpDispatchBuilder::PMADDWD(OpcodeArgs) {
|
||||
|
||||
if (Size == 8) {
|
||||
Size <<= 1;
|
||||
Src1 = _VBitcast(Size, 2, Src1);
|
||||
Src2 = _VBitcast(Size, 2, Src2);
|
||||
}
|
||||
|
||||
auto Src1_L = _VSXTL(Size, 2, Src1); // [15:0 ], [31:16], [32:47 ], [63:48 ]
|
||||
@@ -1954,9 +2098,6 @@ void OpDispatchBuilder::PMULHW(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Res{};
|
||||
if (Size == 8) {
|
||||
Dest = _VBitcast(Size * 2, 2, Dest);
|
||||
Src = _VBitcast(Size * 2, 2, Src);
|
||||
|
||||
// Implementation is more efficient for 8byte registers
|
||||
if (Signed)
|
||||
Res = _VSMull(Size * 2, 2, Dest, Src);
|
||||
@@ -2333,7 +2474,7 @@ void OpDispatchBuilder::VectorRound(OpcodeArgs) {
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Src);
|
||||
auto Result = _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else {
|
||||
@@ -2382,7 +2523,8 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
// The mask is hardcoded to be xmm0 in this instruction
|
||||
OrderedNode *Mask = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
auto Mask = LoadXMMRegister(0);
|
||||
|
||||
// Each element is selected by the high bit of that element size
|
||||
// Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest;
|
||||
//
|
||||
@@ -2617,4 +2759,28 @@ void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VZEROOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto IsVZEROALL = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
if (IsVZEROALL) {
|
||||
// NOTE: Despite the name being VZEROALL, this will still only ever
|
||||
// zero out up to the first 16 registers (even on AVX-512, where we have 32 registers)
|
||||
|
||||
OrderedNode* ZeroVector = _VectorZero(DstSize);
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
StoreXMMRegister(i, ZeroVector);
|
||||
}
|
||||
} else {
|
||||
// Likewise, VZEROUPPER will only ever zero only up to the first 16 registers
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
OrderedNode* Reg = LoadXMMRegister(i);
|
||||
OrderedNode* Dst = _VMov(16, Reg);
|
||||
StoreXMMRegister(i, Dst);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -782,6 +782,12 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -809,7 +815,8 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -854,6 +861,9 @@ void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
|
||||
auto sin = _F80SIN(a);
|
||||
auto cos = _F80COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -900,6 +910,9 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
OrderedNode *data = _VCastFromGPR(16, 8, low);
|
||||
data = _VInsGPR(16, 8, 1, data, high);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -1185,7 +1198,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
|
||||
// reset to default
|
||||
|
||||
@@ -778,6 +778,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -804,7 +810,8 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -831,6 +838,9 @@ void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
|
||||
auto sin = _F64SIN(a);
|
||||
auto cos = _F64COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -871,6 +881,9 @@ void OpDispatchBuilder::X87TANF64(OpcodeArgs) {
|
||||
|
||||
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(one, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -996,7 +1009,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
|
||||
// reset to default
|
||||
|
||||
@@ -266,10 +266,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x17, 1, X86InstInfo{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1E, 1, X86InstInfo{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1F, 1, X86InstInfo{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
{0x48, 8, X86InstInfo{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
@@ -283,8 +283,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -76,7 +76,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -85,7 +85,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -94,7 +94,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -295,7 +295,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
|
||||
{0x20, 4, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x24, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SS", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2B, 1, X86InstInfo{"MOVNTSS", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x2C, 1, X86InstInfo{"CVTTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{0x2D, 1, X86InstInfo{"CVTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
@@ -305,18 +305,18 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 1, X86InstInfo{"RSQRTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x53, 1, X86InstInfo{"RCPSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 1, X86InstInfo{"RSQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x53, 1, X86InstInfo{"RCPSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x54, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSS2SD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSS2SD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5B, 1, X86InstInfo{"CVTTPS2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x60, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x68, 7, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -383,16 +383,16 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSD2SS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5B, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x60, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
|
||||
+51
-51
@@ -17,23 +17,23 @@ void InitializeVEXTables() {
|
||||
static constexpr U16U8InfoStruct VEXTable[] = {
|
||||
// Map 0 (Reserved)
|
||||
// VEX Map 1
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x10), 1, X86InstInfo{"VMODUPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x10), 1, X86InstInfo{"VMOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x10), 1, X86InstInfo{"VMOVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x10), 1, X86InstInfo{"VMOVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x11), 1, X86InstInfo{"VMOVUPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x11), 1, X86InstInfo{"VMODUPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x11), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x11), 1, X86InstInfo{"VMOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x11), 1, X86InstInfo{"VMOVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x11), 1, X86InstInfo{"VMOVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x13), 1, X86InstInfo{"VMOVLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x13), 1, X86InstInfo{"VMOVLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x13), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x13), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x14), 1, X86InstInfo{"VUNPCKLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x14), 1, X86InstInfo{"VUNPCKLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -41,12 +41,12 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b00, 0x15), 1, X86InstInfo{"VUNPCKHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x15), 1, X86InstInfo{"VUNPCKHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x16), 1, X86InstInfo{"VMOVHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x16), 1, X86InstInfo{"VMOVHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x16), 1, X86InstInfo{"VMOVSHDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x16), 1, X86InstInfo{"VMOVHPS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x16), 1, X86InstInfo{"VMOVHPD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x16), 1, X86InstInfo{"VMOVSHDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x17), 1, X86InstInfo{"VMOVHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x17), 1, X86InstInfo{"VMOVHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x17), 1, X86InstInfo{"VMOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x17), 1, X86InstInfo{"VMOVHPD", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x50), 1, X86InstInfo{"VMOVMSKPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x50), 1, X86InstInfo{"VMOVMSKPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -62,17 +62,17 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b00, 0x53), 1, X86InstInfo{"VRCPPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x53), 1, X86InstInfo{"VRCPSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x54), 1, X86InstInfo{"VANDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x54), 1, X86InstInfo{"VANDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x54), 1, X86InstInfo{"VANDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x54), 1, X86InstInfo{"VANDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x55), 1, X86InstInfo{"VANDNPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x55), 1, X86InstInfo{"VANDNPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x55), 1, X86InstInfo{"VANDNPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x55), 1, X86InstInfo{"VANDNPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x56), 1, X86InstInfo{"VORPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x56), 1, X86InstInfo{"VORPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x56), 1, X86InstInfo{"VORPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x56), 1, X86InstInfo{"VORPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x57), 1, X86InstInfo{"VXORPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x57), 1, X86InstInfo{"VDORPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x57), 1, X86InstInfo{"VXORPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x57), 1, X86InstInfo{"VXORPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x60), 1, X86InstInfo{"VPUNPCKLBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x61), 1, X86InstInfo{"VPUNPCKLWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -95,7 +95,7 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x75), 1, X86InstInfo{"VPCMPEQW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x76), 1, X86InstInfo{"VPCMPEQD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x77), 1, X86InstInfo{"VZERO*", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x77), 1, X86InstInfo{"VZERO*", TYPE_INST, GenFlagsDstSize(SIZE_128BIT), 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xC2), 1, X86InstInfo{"VCMPccPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xC2), 1, X86InstInfo{"VCMPccPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -112,17 +112,17 @@ void InitializeVEXTables() {
|
||||
// This table doesn't state which VEX.pp is for which instruction
|
||||
// XXX: Confirm all the above encoding opcodes
|
||||
|
||||
{OPD(1, 0b00, 0x28), 1, X86InstInfo{"VMOVAPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x28), 1, X86InstInfo{"VMOVAPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x28), 1, X86InstInfo{"VMOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x28), 1, X86InstInfo{"VMOVAPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x29), 1, X86InstInfo{"VMOVAPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x29), 1, X86InstInfo{"VMOVAPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x29), 1, X86InstInfo{"VMOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x29), 1, X86InstInfo{"VMOVAPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, X86InstInfo{"VCVTSI2SS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2A), 1, X86InstInfo{"VCVTSI2SD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, X86InstInfo{"VMOVNTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2B), 1, X86InstInfo{"VMOVNTPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x2B), 1, X86InstInfo{"VMOVNTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2B), 1, X86InstInfo{"VMOVNTPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2C), 1, X86InstInfo{"VCVTTSS2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2C), 1, X86InstInfo{"VCVTTSD2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -136,8 +136,8 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b00, 0x2F), 1, X86InstInfo{"VUCOMISS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2F), 1, X86InstInfo{"VUCOMISD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, X86InstInfo{"VADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x58), 1, X86InstInfo{"VADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x58), 1, X86InstInfo{"VADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x58), 1, X86InstInfo{"VADDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x58), 1, X86InstInfo{"VADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x58), 1, X86InstInfo{"VADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -179,8 +179,8 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x6D), 1, X86InstInfo{"VPUNPCKHQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6E), 1, X86InstInfo{"VMOV*", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x6F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x6F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x6F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7C), 1, X86InstInfo{"VHADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7C), 1, X86InstInfo{"VHADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -188,11 +188,11 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x7D), 1, X86InstInfo{"VHSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7D), 1, X86InstInfo{"VHSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7E), 1, X86InstInfo{"VMOV*", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7E), 1, X86InstInfo{"VMOVQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7E), 1, X86InstInfo{"VMOV*", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7E), 1, X86InstInfo{"VMOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
{OPD(1, 0b01, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
@@ -205,19 +205,19 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0xD1), 1, X86InstInfo{"VPSRLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD2), 1, X86InstInfo{"VPSRLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD3), 1, X86InstInfo{"VPSRLQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD4), 1, X86InstInfo{"VPADDQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD4), 1, X86InstInfo{"VPADDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD5), 1, X86InstInfo{"VPMULLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD6), 1, X86InstInfo{"VMOVQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD6), 1, X86InstInfo{"VMOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD7), 1, X86InstInfo{"VPMOVMSKB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xD8), 1, X86InstInfo{"VPSUBUSB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD9), 1, X86InstInfo{"VPSUBUSW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDA), 1, X86InstInfo{"VPMINUB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDB), 1, X86InstInfo{"VPAND", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDB), 1, X86InstInfo{"VPAND", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDC), 1, X86InstInfo{"VPADDUSB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDD), 1, X86InstInfo{"VPADDUSW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDE), 1, X86InstInfo{"VPMAXUB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDF), 1, X86InstInfo{"VPANDN", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDF), 1, X86InstInfo{"VPANDN", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE0), 1, X86InstInfo{"VPAVGB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE1), 1, X86InstInfo{"VPSRAW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -230,16 +230,16 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b10, 0xE6), 1, X86InstInfo{"VCVTDQ2PD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xE6), 1, X86InstInfo{"VCVTPD2DQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE7), 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE8), 1, X86InstInfo{"VPSUBSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE9), 1, X86InstInfo{"VPSUBSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEA), 1, X86InstInfo{"VPMINSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEB), 1, X86InstInfo{"VPOR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEA), 1, X86InstInfo{"VPMINSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEB), 1, X86InstInfo{"VPOR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEC), 1, X86InstInfo{"VPADDSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xED), 1, X86InstInfo{"VPADDSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEE), 1, X86InstInfo{"VPMAXSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEF), 1, X86InstInfo{"VPXOR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEE), 1, X86InstInfo{"VPMAXSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEF), 1, X86InstInfo{"VPXOR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b11, 0xF0), 1, X86InstInfo{"VLDDQU", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -255,9 +255,9 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0xF9), 1, X86InstInfo{"VPSUBW", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFA), 1, X86InstInfo{"VPSUBD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFB), 1, X86InstInfo{"VPSUBQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFC), 1, X86InstInfo{"VPADDB", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFD), 1, X86InstInfo{"VPADDW", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFE), 1, X86InstInfo{"VPADDD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFC), 1, X86InstInfo{"VPADDB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFD), 1, X86InstInfo{"VPADDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFE), 1, X86InstInfo{"VPADDD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
// VEX Map 2
|
||||
{OPD(2, 0b01, 0x00), 1, X86InstInfo{"VPSHUFB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -298,7 +298,7 @@ void InitializeVEXTables() {
|
||||
|
||||
{OPD(2, 0b01, 0x28), 1, X86InstInfo{"VPMULDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x29), 1, X86InstInfo{"VPCMPEQQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2A), 1, X86InstInfo{"VMOVNTDQA", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2A), 1, X86InstInfo{"VMOVNTDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2B), 1, X86InstInfo{"VPACKUSDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2C), 1, X86InstInfo{"VMASKMOVPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2D), 1, X86InstInfo{"VMASKMOVPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
+4
-1
@@ -199,7 +199,7 @@ namespace FEXCore {
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
emit->_StoreContext(GPRSize, IR::GPRClass, emit->_Constant(Entrypoint), offsetof(Core::CPUState, gregs[X86State::REG_R11]));
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
@@ -431,6 +431,9 @@ namespace FEXCore {
|
||||
|
||||
FEX_DEFAULT_VISIBILITY
|
||||
void FinalizeHostTrampolineForGuestFunction(HostToGuestTrampolinePtr* TrampolineAddress, void* HostPacker) {
|
||||
|
||||
if (TrampolineAddress == nullptr) return;
|
||||
|
||||
auto& Trampoline = GetInstanceInfo(TrampolineAddress);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Trampoline.CallCallback == (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
|
||||
+39
-85
@@ -302,10 +302,6 @@
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
"GPR = Mov GPR:$Value": {
|
||||
"DestSize": "GetOpSize(_Value)"
|
||||
},
|
||||
|
||||
"GPR = ExtractElementPair GPRPair:$Pair, u8:$Element": {
|
||||
"Desc": ["Extracts a register for the register pair"],
|
||||
"DestSize": "GetOpSize(_Pair) >> 1"
|
||||
@@ -359,22 +355,26 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContext u8:#ByteSize, RegisterClass:$Class, SSA:$Value, u32:$Offset": {
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -385,7 +385,9 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
"StoreContextIndexed SSA:$Value, GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
|
||||
@@ -397,7 +399,9 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -475,22 +479,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VLoadMemElement u8:#RegisterSize, u8:#ElementSize, FPR:$Value, GPR:$Addr, u8:$Index, u8:$Align{1}": {
|
||||
"Desc": ["Loads an element of size #ElementSize in to $Value from $Addr at $Index"
|
||||
],
|
||||
"OpClass": "Memory",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"VStoreMemElement u8:#RegisterSize, u8:#ElementSize, FPR:$Value, GPR:$Addr, u8:$Index, u8:$Align": {
|
||||
"Desc": ["Stores an element of size #ElementSize from $Value[$Index] to $Addr"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ElementSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"CacheLineClear GPR:$Addr": {
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified",
|
||||
"Only clears the data cachelines. Doesn't do any zeroing"
|
||||
@@ -845,27 +833,25 @@
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))"
|
||||
},
|
||||
"GPR = Extr GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
"Valid LSB range is 0-31 for 32bit and 0-63 for 64bit",
|
||||
"<Size * 2> ConcatValue = $Upper:$Lower",
|
||||
"Result = ConcatValue<LSB+Size - 1: LSB>"
|
||||
]
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
"Valid LSB range is 0-31 for 32bit and 0-63 for 64bit",
|
||||
"<Size * 2> ConcatValue = $Upper:$Lower",
|
||||
"Result = ConcatValue<LSB+Size - 1: LSB>"
|
||||
]
|
||||
},
|
||||
"GPR = PDep GPR:$Input, GPR:$Mask": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
]
|
||||
"Desc": ["Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = PExt GPR:$Input, GPR:$Mask": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
]
|
||||
"Desc": ["Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = LDiv GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
@@ -924,15 +910,6 @@
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
"FPR = SplatVector2 FPR:$Scalar": {
|
||||
"NumElements": "2",
|
||||
"DestSize": "GetOpSize(_Scalar) * 2"
|
||||
},
|
||||
"FPR = SplatVector4 FPR:$Scalar": {
|
||||
"NumElements": "4",
|
||||
"DestSize": "GetOpSize(_Scalar) * 4"
|
||||
},
|
||||
|
||||
"FPR = VMov u8:#RegisterSize, FPR:$Source": {
|
||||
"Desc" : ["Copy vector register",
|
||||
"When Register size is smaller than Source register size,",
|
||||
@@ -941,12 +918,6 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VBitcast u8:#RegisterSize, u8:#ElementSize, FPR:$Source": {
|
||||
"Desc": ["Workaround for issue with LLVM breaking when loading scalar elements to vectors"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VectorZero u8:#RegisterSize": {
|
||||
"Desc": ["Generates a vector zero",
|
||||
"Useful to generate a zero vector without any previous dependencies"
|
||||
@@ -1038,24 +1009,11 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VExtractElement u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"DestSize": "ElementSize"
|
||||
},
|
||||
"FPR = VDupElement u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"Desc": ["Duplicates one element from the source register across the whole register"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSLI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$ByteShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VSRI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$ByteShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VShlI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1089,7 +1047,7 @@
|
||||
},
|
||||
"FPR = VSXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1101,7 +1059,7 @@
|
||||
},
|
||||
"FPR = VUXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Zero extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1124,9 +1082,9 @@
|
||||
},
|
||||
|
||||
"FPR = VRev64 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1320,10 +1278,6 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsScalarElement u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, FPR:$DestVector, FPR:$SrcScalar": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsGPR u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, FPR:$DestVector, GPR:$Src": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1587,9 +1541,9 @@
|
||||
"DestSize": "16"
|
||||
},
|
||||
"GPR = F80Cmp FPR:$X80Src1, FPR:$X80Src2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"FPR = F80BCDLoad FPR:$X80Src": {
|
||||
|
||||
@@ -162,6 +162,12 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
else if (Arg == "FPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRClass};
|
||||
}
|
||||
else if (Arg == "GPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRFixedClass};
|
||||
}
|
||||
else if (Arg == "FPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRFixedClass};
|
||||
}
|
||||
else if (Arg == "GPRPair") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRPairClass};
|
||||
}
|
||||
|
||||
+3
-9
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
@@ -36,15 +37,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineCo
|
||||
|
||||
InsertPass(CreateSyscallOptimization());
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
else {
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
@@ -66,6 +58,8 @@ void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAV
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
bool Changed = false;
|
||||
for (auto const &Pass : Passes) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
|
||||
@@ -21,7 +21,6 @@ std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusiveP
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
|
||||
bool OptimizeSRA,
|
||||
bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
namespace Validation {
|
||||
|
||||
+12
-9
@@ -6,7 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
|
||||
#if defined(_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
//aarch64 heuristics
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
@@ -20,6 +20,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
@@ -45,20 +46,20 @@ uint64_t getMask(IROp_Header* Op) {
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
// very lazy heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; }
|
||||
static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8;
|
||||
}
|
||||
#elif defined(_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
//aarch64 heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { if (width < 32) width = 32; return vixl::aarch64::Assembler::IsImmLogical(imm, width); }
|
||||
static bool IsImmAddSub(uint64_t imm) { return vixl::aarch64::Assembler::IsImmAddSub(imm); }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == AccessSize;
|
||||
}
|
||||
#elif JIT_X86_64
|
||||
// very lazy heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; }
|
||||
static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8;
|
||||
}
|
||||
#else
|
||||
#error No inline constant heuristics for this target
|
||||
#endif
|
||||
@@ -1028,6 +1029,8 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
|
||||
bool ConstProp::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
@@ -22,6 +23,7 @@ private:
|
||||
};
|
||||
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
|
||||
|
||||
+134
-66
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
@@ -76,24 +77,6 @@ namespace {
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // SSE padding in non-AVX case
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
@@ -102,7 +85,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
sizeof(FEXCore::Core::CPUState::rip),
|
||||
},
|
||||
DefaultAccess[0],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -112,62 +95,134 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
|
||||
FEXCore::Core::CPUState::GPR_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[1],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es),
|
||||
sizeof(FEXCore::Core::CPUState::es),
|
||||
offsetof(FEXCore::Core::CPUState, es_idx),
|
||||
sizeof(FEXCore::Core::CPUState::es_idx),
|
||||
},
|
||||
DefaultAccess[2],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs),
|
||||
sizeof(FEXCore::Core::CPUState::cs),
|
||||
offsetof(FEXCore::Core::CPUState, cs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::cs_idx),
|
||||
},
|
||||
DefaultAccess[3],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss),
|
||||
sizeof(FEXCore::Core::CPUState::ss),
|
||||
offsetof(FEXCore::Core::CPUState, ss_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ss_idx),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds),
|
||||
sizeof(FEXCore::Core::CPUState::ds),
|
||||
offsetof(FEXCore::Core::CPUState, ds_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ds_idx),
|
||||
},
|
||||
DefaultAccess[5],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::gs_idx),
|
||||
},
|
||||
DefaultAccess[6],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
sizeof(FEXCore::Core::CPUState::fs),
|
||||
offsetof(FEXCore::Core::CPUState, fs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::fs_idx),
|
||||
},
|
||||
DefaultAccess[7],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad),
|
||||
sizeof(FEXCore::Core::CPUState::_pad),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es_cached),
|
||||
sizeof(FEXCore::Core::CPUState::es_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::cs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ss_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ds_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::gs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::fs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -178,7 +233,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -189,7 +244,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -199,7 +254,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
DefaultAccess[9],
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -210,7 +265,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
|
||||
FEXCore::Core::CPUState::FLAG_SIZE,
|
||||
},
|
||||
DefaultAccess[10],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -221,7 +276,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
|
||||
FEXCore::Core::CPUState::MM_REG_SIZE
|
||||
},
|
||||
DefaultAccess[11],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -233,7 +288,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::gdt[0]),
|
||||
},
|
||||
DefaultAccess[12],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -244,7 +299,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FCW),
|
||||
sizeof(FEXCore::Core::CPUState::FCW),
|
||||
},
|
||||
DefaultAccess[13],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -254,7 +309,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FTW),
|
||||
sizeof(FEXCore::Core::CPUState::FTW),
|
||||
},
|
||||
DefaultAccess[14],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -288,40 +343,55 @@ namespace {
|
||||
ContextClassification->at(Offset).StoreNode = nullptr;
|
||||
};
|
||||
size_t Offset = 0;
|
||||
SetAccess(Offset++, DefaultAccess[0]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[1]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[2]);
|
||||
SetAccess(Offset++, DefaultAccess[3]);
|
||||
SetAccess(Offset++, DefaultAccess[4]);
|
||||
SetAccess(Offset++, DefaultAccess[5]);
|
||||
SetAccess(Offset++, DefaultAccess[6]);
|
||||
SetAccess(Offset++, DefaultAccess[7]);
|
||||
// Segment indexes
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
// Segments
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad2
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[8]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
if (!SupportsAVX) {
|
||||
SetAccess(Offset++, DefaultAccess[9]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[10]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[11]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[12]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
SetAccess(Offset++, DefaultAccess[14]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
@@ -449,7 +519,6 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa29 i128 = VBitcast %ssa26 i128
|
||||
*
|
||||
* eg.
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
@@ -462,13 +531,11 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* eg.
|
||||
* (%%ssa189) StoreContext %ssa188 i128, 0x10, 0xa0
|
||||
* %ssa190 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa191 i128 = VBitcast %ssa188 i128
|
||||
* %ssa192 i128 = VAdd %ssa191 i128, %ssa190 i128, 0x10, 0x4
|
||||
* %ssa192 i128 = VAdd %ssa188 i128, %ssa190 i128, 0x10, 0x4
|
||||
* (%%ssa193) StoreContext %ssa192 i128, 0x10, 0xa0
|
||||
* Converts to
|
||||
* %ssa173 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa174 i128 = VBitcast %ssa172 i128
|
||||
* %ssa175 i128 = VAdd %ssa174 i128, %ssa173 i128, 0x10, 0x4
|
||||
* %ssa175 i128 = VAdd %ssa172 i128, %ssa173 i128, 0x10, 0x4
|
||||
* (%%ssa176) StoreContext %ssa175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
@@ -698,6 +765,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
|
||||
// XXX: We don't do cross-block optimizations yet
|
||||
//CalculateControlFlowInfo(IREmit);
|
||||
bool Changed = false;
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
@@ -154,6 +155,8 @@ struct Info {
|
||||
*
|
||||
*/
|
||||
bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
std::unordered_map<OrderedNode*, Info> InfoMap;
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
@@ -52,6 +53,8 @@ IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
}
|
||||
|
||||
bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRCompaction");
|
||||
|
||||
LocalBuilder.ReownOrClaimBuffer();
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
@@ -32,6 +33,8 @@ IRValidation::~IRValidation() {
|
||||
}
|
||||
|
||||
bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRValidation");
|
||||
|
||||
bool HadError = false;
|
||||
bool HadWarning = false;
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -53,6 +54,8 @@ bool LongDivideEliminationPass::IsSextOp(IREmitter *IREmit, OrderedNodeWrapper L
|
||||
}
|
||||
|
||||
bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::LDE");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
@@ -24,6 +25,8 @@ public:
|
||||
};
|
||||
|
||||
bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::PHIValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <deque>
|
||||
@@ -191,6 +191,8 @@ private:
|
||||
bool RAValidation::Run(IREmitter *IREmit) {
|
||||
if (!Manager->HasPass("RA")) return false;
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
BlockExitState.clear();
|
||||
// BlocksToVisit will already be empty
|
||||
|
||||
+4
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <array>
|
||||
@@ -32,6 +34,8 @@ public:
|
||||
*
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
std::array<OrderedNode*, 32> LastValidFlagStores{};
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -15,6 +15,8 @@ $end_info$
|
||||
#include <FEXCore/Utils/BucketList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -1527,6 +1529,7 @@ namespace {
|
||||
}
|
||||
|
||||
bool ConstrainedRAPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RA");
|
||||
bool Changed = false;
|
||||
|
||||
auto IR = IREmit->ViewIR();
|
||||
|
||||
-128
@@ -1,128 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Replaces Load/StoreContext with Load/StoreReg for SRA regs
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class StaticRegisterAllocationPass final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit StaticRegisterAllocationPass(bool SupportsAVX_) : SupportsAVX{SupportsAVX_} {}
|
||||
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
|
||||
private:
|
||||
bool SupportsAVX;
|
||||
|
||||
bool IsStaticAllocGpr(uint32_t Offset, RegisterClassType Class) const {
|
||||
const auto begin = offsetof(Core::CPUState, gregs[0]);
|
||||
const auto end = offsetof(Core::CPUState, gregs[16]);
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto reg = (Offset - begin) / Core::CPUState::GPR_REG_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(Class.Val == IR::GPRClass.Val, "unexpected Class {}", Class);
|
||||
|
||||
// 0..15 -> 16 in total
|
||||
return reg < Core::CPUState::NUM_GPRS;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool IsStaticAllocFpr(uint32_t Offset, RegisterClassType Class, bool AllowGpr) const {
|
||||
const auto [begin, end] = [this]() -> std::pair<ptrdiff_t, ptrdiff_t> {
|
||||
if (SupportsAVX) {
|
||||
return {
|
||||
offsetof(Core::CPUState, xmm.avx.data[0][0]),
|
||||
offsetof(Core::CPUState, xmm.avx.data[16][0]),
|
||||
};
|
||||
} else {
|
||||
return {
|
||||
offsetof(Core::CPUState, xmm.sse.data[0][0]),
|
||||
offsetof(Core::CPUState, xmm.sse.data[16][0]),
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto size = SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto reg = (Offset - begin) / size;
|
||||
LOGMAN_THROW_AA_FMT(Class.Val == IR::FPRClass.Val || (AllowGpr && Class.Val == IR::GPRClass.Val), "unexpected Class {}, AllowGpr {}", Class, AllowGpr);
|
||||
|
||||
// 0..15 -> 16 in total
|
||||
return reg < Core::CPUState::NUM_XMMS;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This pass replaces Load/Store Context with Load/Store Register for Statically Mapped registers. It also does some validation.
|
||||
*
|
||||
*/
|
||||
bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
|
||||
if (IsStaticAllocGpr(Op->Offset, Op->Class) || IsStaticAllocFpr(Op->Offset, Op->Class, true)) {
|
||||
auto GeneralClass = Op->Class;
|
||||
if (IsStaticAllocFpr(Op->Offset, GeneralClass, true) && GeneralClass == GPRClass) {
|
||||
GeneralClass = FPRClass;
|
||||
}
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
OrderedNode *sraReg = IREmit->_LoadRegister(false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
if (GeneralClass != Op->Class) {
|
||||
sraReg = IREmit->_VExtractToGPR(Op->Header.Size, Op->Header.Size, sraReg, 0);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, sraReg);
|
||||
}
|
||||
} if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
|
||||
if (IsStaticAllocGpr(Op->Offset, Op->Class) || IsStaticAllocFpr(Op->Offset, Op->Class, true)) {
|
||||
auto val = IREmit->UnwrapNode(Op->Value);
|
||||
|
||||
auto GeneralClass = Op->Class;
|
||||
if (IsStaticAllocFpr(Op->Offset, GeneralClass, true) && GeneralClass == GPRClass) {
|
||||
val = IREmit->_VCastFromGPR(Op->Header.Size, Op->Header.Size, val);
|
||||
GeneralClass = FPRClass;
|
||||
}
|
||||
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
IREmit->_StoreRegister(val, false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX) {
|
||||
return std::make_unique<StaticRegisterAllocationPass>(SupportsAVX);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -23,6 +24,8 @@ public:
|
||||
};
|
||||
|
||||
bool SyscallOptimization::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
@@ -36,6 +37,8 @@ public:
|
||||
};
|
||||
|
||||
bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ValueDominanceValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
+38
-7
@@ -45,6 +45,8 @@ namespace FEXCore::Allocator {
|
||||
FREE_Hook free {::free};
|
||||
#endif
|
||||
|
||||
uint64_t HostVASize{};
|
||||
|
||||
using GLIBC_MALLOC_Hook = void*(*)(size_t, const void *caller);
|
||||
using GLIBC_REALLOC_Hook = void*(*)(void*, size_t, const void *caller);
|
||||
using GLIBC_FREE_Hook = void(*)(void*, const void *caller);
|
||||
@@ -97,6 +99,10 @@ namespace FEXCore::Allocator {
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
|
||||
if (HostVASize) {
|
||||
return HostVASize;
|
||||
}
|
||||
|
||||
static constexpr std::array<uintptr_t, 7> TLBSizes = {
|
||||
57,
|
||||
52,
|
||||
@@ -127,6 +133,7 @@ namespace FEXCore::Allocator {
|
||||
};
|
||||
|
||||
if (Find(Size)) {
|
||||
HostVASize = Bits;
|
||||
return Bits;
|
||||
}
|
||||
}
|
||||
@@ -138,8 +145,10 @@ namespace FEXCore::Allocator {
|
||||
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
|
||||
|
||||
std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
void * const StackLocation = alloca(0);
|
||||
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(StackLocation);
|
||||
std::vector<MemoryRegion> Regions;
|
||||
|
||||
|
||||
int MapsFD = open("/proc/self/maps", O_RDONLY);
|
||||
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
|
||||
|
||||
@@ -148,6 +157,8 @@ namespace FEXCore::Allocator {
|
||||
uintptr_t RegionBegin = 0;
|
||||
uintptr_t RegionEnd = 0;
|
||||
|
||||
uintptr_t PreviousMapEnd = 0;
|
||||
|
||||
char Buffer[2048];
|
||||
const char *Cursor;
|
||||
ssize_t Remaining = 0;
|
||||
@@ -155,7 +166,7 @@ namespace FEXCore::Allocator {
|
||||
for(;;) {
|
||||
|
||||
if (Remaining == 0) {
|
||||
do {
|
||||
do {
|
||||
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
|
||||
} while ( Remaining == -1 && errno == EAGAIN);
|
||||
|
||||
@@ -165,8 +176,8 @@ namespace FEXCore::Allocator {
|
||||
if (Remaining == 0 && State == ParseBegin) {
|
||||
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = End;
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = End;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
@@ -202,9 +213,12 @@ namespace FEXCore::Allocator {
|
||||
if (c == '-') {
|
||||
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
// Store the location we are going to map.
|
||||
PreviousMapEnd = MapEnd;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
if (MapEnd > MapBegin) {
|
||||
@@ -218,6 +232,7 @@ namespace FEXCore::Allocator {
|
||||
|
||||
Regions.push_back({(void*)MapBegin, MapSize});
|
||||
}
|
||||
|
||||
RegionBegin = 0;
|
||||
RegionEnd = 0;
|
||||
State = ParseEnd;
|
||||
@@ -233,6 +248,22 @@ namespace FEXCore::Allocator {
|
||||
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
State = ScanEnd;
|
||||
|
||||
// If the previous map's ending and the region we just parsed overlap the stack then we need to save the stack mapping.
|
||||
// Otherwise we will have severely limited stack size which crashes quickly.
|
||||
if (PreviousMapEnd <= StackLocation_u64 && RegionEnd > StackLocation_u64) {
|
||||
auto BelowStackRegion = Regions.back();
|
||||
LOGMAN_THROW_AA_FMT(reinterpret_cast<uint64_t>(BelowStackRegion.Ptr) + BelowStackRegion.Size == PreviousMapEnd,
|
||||
"This needs to match");
|
||||
|
||||
// Allocate the region under the stack as READ | WRITE so the stack can still grow
|
||||
auto Alloc = mmap(BelowStackRegion.Ptr, BelowStackRegion.Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", BelowStackRegion.Ptr, BelowStackRegion.Size);
|
||||
LogMan::Throw::AFmt(Alloc == BelowStackRegion.Ptr, "mmap({},{:x}) returned {} instead of {:x}", Alloc, BelowStackRegion.Ptr);
|
||||
|
||||
Regions.pop_back();
|
||||
}
|
||||
continue;
|
||||
} else {
|
||||
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
|
||||
|
||||
+83
-102
@@ -69,11 +69,14 @@ namespace Alloc::OSAllocator {
|
||||
struct LiveVMARegion {
|
||||
ReservedVMARegion *SlabInfo;
|
||||
uint64_t FreeSpace{};
|
||||
uint64_t NumManagedPages{};
|
||||
uint32_t LastPageAllocation{};
|
||||
bool HadMunmap{};
|
||||
|
||||
// Align UsedPages so it pads to the next page.
|
||||
// Necessary to take advantage of madvise zero page pooling.
|
||||
alignas(4096) FEXCore::FlexBitSet<uint64_t> UsedPages;
|
||||
using FlexBitElementType = uint64_t;
|
||||
alignas(4096) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
|
||||
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
|
||||
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
|
||||
@@ -85,8 +88,8 @@ namespace Alloc::OSAllocator {
|
||||
// 0x100'0000 Pages
|
||||
// 1 bit per page for tracking means 0x20'0000 (Pages / 8) bytes of flex space
|
||||
// Which is 2MB of tracking
|
||||
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(uint64_t);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<uint64_t>::Size(NumElements);
|
||||
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::Size(NumElements);
|
||||
}
|
||||
|
||||
static void InitializeVMARegionUsed(LiveVMARegion *Region, size_t AdditionalSize) {
|
||||
@@ -95,19 +98,21 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
|
||||
|
||||
size_t NumPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
|
||||
size_t NumManagedPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
|
||||
size_t ManagedSize = NumManagedPages << FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
// Use madvise to set the full tracking region to zero.
|
||||
// This ensures unused pages are zero, while not having the backing pages consuming memory.
|
||||
::madvise(Region->UsedPages.Memory + (NumPages * 4096), (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - (NumPages * 4096), MADV_DONTNEED);
|
||||
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - ManagedSize, MADV_DONTNEED);
|
||||
|
||||
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
|
||||
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
|
||||
::madvise(Region->UsedPages.Memory, NumPages * 4096, MADV_WILLNEED);
|
||||
::madvise(Region->UsedPages.Memory, ManagedSize, MADV_WILLNEED);
|
||||
|
||||
// Set our reserved pages
|
||||
Region->UsedPages.MemSet(NumPages);
|
||||
Region->LastPageAllocation = NumPages;
|
||||
Region->UsedPages.MemSet(NumManagedPages);
|
||||
Region->LastPageAllocation = NumManagedPages;
|
||||
Region->NumManagedPages = NumManagedPages;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -129,6 +134,7 @@ namespace Alloc::OSAllocator {
|
||||
ReservedVMARegion *ReservedRegion = *ReservedIterator;
|
||||
|
||||
ReservedRegions->erase(ReservedIterator);
|
||||
|
||||
// mprotect the new region we've allocated
|
||||
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FHU::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
|
||||
@@ -152,6 +158,9 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
// 32-bit old kernel workarounds
|
||||
std::vector<FEXCore::Allocator::MemoryRegion> Steal32BitIfOldKernel();
|
||||
|
||||
void AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges);
|
||||
LiveVMARegion *FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd);
|
||||
};
|
||||
|
||||
void OSAllocator_64Bit::DetermineVASize() {
|
||||
@@ -167,6 +176,42 @@ void OSAllocator_64Bit::DetermineVASize() {
|
||||
UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::LiveVMARegion *OSAllocator_64Bit::FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd) {
|
||||
LiveVMARegion *LiveRegion{};
|
||||
|
||||
// Check active slabs to see if we can fit this
|
||||
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin &&
|
||||
Addr < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Couldn't find an active region that fit
|
||||
// Check reserved regions
|
||||
if (!LiveRegion) {
|
||||
// Didn't have a slab that fit this range
|
||||
// Check our reserved regions to see if we have one that fits
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
ReservedVMARegion *ReservedRegion = *it;
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base &&
|
||||
AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return LiveRegion;
|
||||
}
|
||||
|
||||
void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
if (addr != 0 &&
|
||||
addr < reinterpret_cast<void*>(LOWER_BOUND)) {
|
||||
@@ -205,41 +250,13 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
LiveVMARegion *LiveRegion{};
|
||||
|
||||
if (Fixed || Addr != 0) {
|
||||
// Check active slabs to see if we can fit this
|
||||
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin &&
|
||||
Addr < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Couldn't find an active region that fit
|
||||
// Check reserved regions
|
||||
if (!LiveRegion) {
|
||||
// Didn't have a slab that fit this range
|
||||
// Check our reserved regions to see if we have one that fits
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
ReservedVMARegion *ReservedRegion = *it;
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base &&
|
||||
AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
LiveRegion = FindLiveRegionForAddress(Addr, AddrEnd);
|
||||
}
|
||||
|
||||
again:
|
||||
|
||||
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion *Region, uint64_t length, int prot, int flags, int fd, off_t offset, uint64_t StartingPosition = 0) -> std::pair<LiveVMARegion*, void*> {
|
||||
uint64_t AllocatedPage{};
|
||||
uint64_t AllocatedPage{~0ULL};
|
||||
uint64_t NumberOfPages = length >> FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
if (Region->FreeSpace >= length) {
|
||||
@@ -249,72 +266,29 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
: Region->LastPageAllocation;
|
||||
size_t RegionNumberOfPages = Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
// Backward scan
|
||||
// We need to do a backward scan first to fill any holes
|
||||
// Otherwise we will very quickly run out of VMA regions (65k maximum)
|
||||
for (size_t CurrentPage = LastAllocation;
|
||||
CurrentPage >= NumberOfPages;) {
|
||||
size_t Remaining = NumberOfPages;
|
||||
assert(Remaining <= CurrentPage);
|
||||
|
||||
while (Remaining) {
|
||||
if (Region->UsedPages[CurrentPage - Remaining]) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
if (Region->HadMunmap) {
|
||||
// Backward scan
|
||||
// We need to do a backward scan first to fill any holes
|
||||
// Otherwise we will very quickly run out of VMA regions (65k maximum)
|
||||
auto SearchResult = Region->UsedPages.BackwardScanForRange<true>(LastAllocation, NumberOfPages, Region->NumManagedPages);
|
||||
|
||||
if (Remaining) {
|
||||
// Didn't find a slab range
|
||||
CurrentPage -= Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
CurrentPage -= NumberOfPages;
|
||||
AllocatedPage = SearchResult.FoundElement;
|
||||
|
||||
// Keep scanning backwards to not introduce ANOTHER gap
|
||||
while (CurrentPage >= 1) {
|
||||
if (Region->UsedPages[CurrentPage - 1]) {
|
||||
// Found a used page, we can leave now
|
||||
break;
|
||||
}
|
||||
--CurrentPage;
|
||||
}
|
||||
AllocatedPage = CurrentPage;
|
||||
break;
|
||||
// If we didn't even have a one page free in the backward search, then unclaim HadMunmap.
|
||||
// Switching over to default forward search.
|
||||
if (SearchResult.FoundElement == ~0ULL && !SearchResult.FoundHole) {
|
||||
Region->HadMunmap = false;
|
||||
}
|
||||
}
|
||||
|
||||
// Foward Scan
|
||||
if (AllocatedPage == 0) {
|
||||
for (size_t CurrentPage = LastAllocation;
|
||||
CurrentPage < (RegionNumberOfPages - NumberOfPages);) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = NumberOfPages;
|
||||
|
||||
assert((CurrentPage + Remaining - 1) < RegionNumberOfPages);
|
||||
while (Remaining) {
|
||||
if (Region->UsedPages[CurrentPage + Remaining - 1]) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// Didn't find a slab range
|
||||
CurrentPage += Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
AllocatedPage = CurrentPage;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (AllocatedPage == ~0ULL) {
|
||||
auto SearchResult = Region->UsedPages.ForwardScanForRange<true>(LastAllocation, NumberOfPages, RegionNumberOfPages);
|
||||
AllocatedPage = SearchResult.FoundElement;
|
||||
}
|
||||
|
||||
if (AllocatedPage) {
|
||||
if (AllocatedPage != ~0ULL) {
|
||||
AllocatedOffset = Region->SlabInfo->Base + AllocatedPage * FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// We need to setup protections for this
|
||||
@@ -497,6 +471,8 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
// This will let us more quickly fill holes
|
||||
(*it)->LastPageAllocation = std::min((*it)->LastPageAllocation, SlabPageBegin);
|
||||
|
||||
(*it)->HadMunmap = true;
|
||||
|
||||
// XXX: Move region back to reserved list
|
||||
return 0;
|
||||
}
|
||||
@@ -537,12 +513,7 @@ std::vector<FEXCore::Allocator::MemoryRegion> OSAllocator_64Bit::Steal32BitIfOld
|
||||
return FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND_32, UPPER_BOUND_32);
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
DetermineVASize();
|
||||
auto LowMem = Steal32BitIfOldKernel();
|
||||
|
||||
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
|
||||
|
||||
void OSAllocator_64Bit::AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges) {
|
||||
for (auto [Ptr, AllocationSize]: Ranges) {
|
||||
if (!ObjectAlloc) {
|
||||
auto MaxSize = std::min(size_t(64) * 1024 * 1024, AllocationSize);
|
||||
@@ -564,12 +535,22 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
ReservedVMARegion *Region = ObjectAlloc->new_construct<ReservedVMARegion>();
|
||||
Region->Base = reinterpret_cast<uint64_t>(Ptr);
|
||||
Region->RegionSize = AllocationSize;
|
||||
ReservedRegions->emplace_back(Region);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
DetermineVASize();
|
||||
auto LowMem = Steal32BitIfOldKernel();
|
||||
|
||||
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
|
||||
|
||||
AllocateMemoryRegions(Ranges);
|
||||
|
||||
FEXCore::Allocator::ReclaimMemoryRegion(LowMem);
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -38,6 +39,109 @@ struct FlexBitSet final {
|
||||
memset(Memory, 0xFF, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
}
|
||||
|
||||
// Range scanning results
|
||||
struct BitsetScanResults {
|
||||
// Which element was found. ~0ULL if not found.
|
||||
size_t FoundElement;
|
||||
// During the scan, found a hole in the allocations that didn't fit.
|
||||
bool FoundHole;
|
||||
};
|
||||
|
||||
// TODO: Make {Forward,Backward}ScanForRange faster
|
||||
// Currently these functions test a single bit at a time, which is fairly costly.
|
||||
// The compiler emits a full element load per iteration, wasting a bunch of time on loads.
|
||||
// If we change these functions to have a pre-amble and post-amble to align the primary loop to the element size then this can go significantly
|
||||
// faster.
|
||||
//
|
||||
// Once the element scanning is aligned to the element size, we can then use native count leading zero(CLZ) and count trailing zero(CTZ)
|
||||
// instructions on a full element to scan uint64_t elements per loop iteration.
|
||||
|
||||
// Implementation details:
|
||||
// Template argument WantUnset
|
||||
// Used to determine if the desired range is for set or unset ranges.
|
||||
// Typically `WantUnset` should be true. Used for finding a unset range inside of a range will set elements.
|
||||
//
|
||||
// @param BeginningElement - The first element in the set to start scanning from.
|
||||
// @param ElementCount - How many elements to find a range for fitting.
|
||||
// @param MinimumElement - Minimum element in the set to search to
|
||||
//
|
||||
// @return The scan results
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults BackwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t MinimumElement) {
|
||||
bool FoundHole {};
|
||||
for (size_t CurrentPage = BeginningElement;
|
||||
CurrentPage >= (MinimumElement + ElementCount);) {
|
||||
size_t Remaining = ElementCount;
|
||||
LOGMAN_THROW_AA_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentPage - Remaining) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// If we found at least one Element hole then track that
|
||||
if (Remaining != ElementCount) {
|
||||
FoundHole = true;
|
||||
}
|
||||
|
||||
// Didn't find a slab range
|
||||
CurrentPage -= Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults{CurrentPage - ElementCount, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
return BitsetScanResults {~0ULL, FoundHole};
|
||||
}
|
||||
|
||||
// @param BeginningElement - The first element in the set to start scanning from.
|
||||
// @param ElementCount - How many elements to find a range for fitting.
|
||||
// @param ElementsInSet - How many elements are in the full set.
|
||||
//
|
||||
// @return The scan results
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults ForwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t ElementsInSet) {
|
||||
bool FoundHole {};
|
||||
|
||||
for (size_t CurrentElement = BeginningElement;
|
||||
CurrentElement < (ElementsInSet - ElementCount);) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = ElementCount;
|
||||
|
||||
LOGMAN_THROW_AA_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// If we found at least one Element hole then track that
|
||||
if (Remaining != ElementCount) {
|
||||
FoundHole = true;
|
||||
}
|
||||
|
||||
// Didn't find a slab range
|
||||
CurrentElement += Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults {CurrentElement, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
return BitsetScanResults {~0ULL, FoundHole};
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
bool operator[](size_t Element) const {
|
||||
|
||||
+116
@@ -0,0 +1,116 @@
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <linux/magic.h>
|
||||
#include <string>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/vfs.h>
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#define BACKEND_OFF 0
|
||||
#define BACKEND_GPUVIS 1
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
namespace FEXCore::Profiler {
|
||||
ProfilerBlock::ProfilerBlock(std::string_view const Format)
|
||||
: DurationBegin {GetTime()}
|
||||
, Format {Format} {
|
||||
}
|
||||
|
||||
ProfilerBlock::~ProfilerBlock() {
|
||||
auto Duration = GetTime() - DurationBegin;
|
||||
TraceObject(Format, Duration);
|
||||
}
|
||||
}
|
||||
|
||||
namespace GPUVis {
|
||||
// ftrace FD for writing trace data.
|
||||
// Needs to be a raw FD since we hold this open for the entire application execution.
|
||||
static int TraceFD {-1};
|
||||
|
||||
// Need to search the paths to find the real trace path
|
||||
static std::array<char const*, 2> TraceFSDirectories {
|
||||
"/sys/kernel/tracing",
|
||||
"/sys/kernel/debug/tracing",
|
||||
};
|
||||
|
||||
static bool IsTraceFS(char const* Path) {
|
||||
struct statfs stat;
|
||||
if (statfs(Path, &stat)) {
|
||||
return false;
|
||||
}
|
||||
return stat.f_type == TRACEFS_MAGIC;
|
||||
}
|
||||
|
||||
void Init() {
|
||||
for (auto Path : TraceFSDirectories) {
|
||||
if (IsTraceFS(Path)) {
|
||||
std::string FilePath = fmt::format("{}/trace_marker", Path);
|
||||
TraceFD = open(FilePath.c_str(), O_WRONLY | O_CLOEXEC);
|
||||
if (TraceFD != -1) {
|
||||
// Opened TraceFD, early exit
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
if (TraceFD != -1) {
|
||||
close(TraceFD);
|
||||
TraceFD = -1;
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
if (TraceFD != -1) {
|
||||
// Print the duration as something that began negative duration ago
|
||||
std::string Event = fmt::format("{} (lduration=-{})\n", Format, Duration);
|
||||
write(TraceFD, Event.c_str(), Event.size());
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
if (TraceFD != -1) {
|
||||
std::string Event = fmt::format("{}\n", Format);
|
||||
write(TraceFD, Format.data(), Format.size());
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
#error Unknown profiler backend
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
void Init() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::Init();
|
||||
#endif
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::Shutdown();
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format, Duration);
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format);
|
||||
#endif
|
||||
|
||||
}
|
||||
#endif
|
||||
}
|
||||
+3
-1
@@ -74,7 +74,9 @@ namespace Handler {
|
||||
LAYER_GLOBAL_MAIN, ///< /usr/share/fex-emu/Config.json by default
|
||||
LAYER_MAIN,
|
||||
LAYER_ARGUMENTS,
|
||||
LAYER_GLOBAL_STEAM_APP,
|
||||
LAYER_GLOBAL_APP,
|
||||
LAYER_LOCAL_STEAM_APP,
|
||||
LAYER_LOCAL_APP,
|
||||
LAYER_ENVIRONMENT,
|
||||
LAYER_TOP,
|
||||
@@ -272,7 +274,7 @@ namespace Type {
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
|
||||
@@ -122,6 +122,10 @@ namespace CPU {
|
||||
bool IsAddressInCodeBuffer(uintptr_t Address) const;
|
||||
|
||||
protected:
|
||||
// Max spill slot size in bytes. We need at most 32 bytes
|
||||
// to be able to handle a 256-bit vector store to a slot.
|
||||
constexpr static uint32_t MaxSpillSlotSize = 32;
|
||||
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
|
||||
size_t InitialCodeSize, MaxCodeSize;
|
||||
|
||||
+18
-7
@@ -28,9 +28,16 @@ namespace FEXCore::Core {
|
||||
|
||||
uint64_t rip; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16];
|
||||
uint16_t es, cs, ss, ds;
|
||||
uint64_t gs;
|
||||
uint64_t fs;
|
||||
// Raw segment register indexes
|
||||
uint16_t es_idx, cs_idx, ss_idx, ds_idx;
|
||||
uint16_t gs_idx, fs_idx;
|
||||
uint16_t _pad[2];
|
||||
|
||||
// Segment registers holding base addresses
|
||||
uint32_t es_cached, cs_cached, ss_cached, ds_cached;
|
||||
uint64_t gs_cached;
|
||||
uint64_t fs_cached;
|
||||
uint64_t _pad2[1];
|
||||
XMMRegs xmm;
|
||||
uint8_t flags[48];
|
||||
uint64_t mm[8][2];
|
||||
@@ -211,12 +218,13 @@ namespace FEXCore::Core {
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
|
||||
struct SynchronousFaultDataStruct {
|
||||
struct alignas(8) SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint8_t Signal;
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
uint8_t TrapNo;
|
||||
uint8_t si_code;
|
||||
uint16_t err_code;
|
||||
uint32_t _pad : 16;
|
||||
} SynchronousFaultData;
|
||||
|
||||
InternalThreadState* Thread;
|
||||
@@ -230,6 +238,9 @@ namespace FEXCore::Core {
|
||||
static_assert(offsetof(CpuStateFrame, Pointers) + sizeof(CpuStateFrame::Pointers) <= 32760, "JITPointers maximum pointer needs to be less than architecture maximum 32768");
|
||||
|
||||
static_assert(std::is_standard_layout<CpuStateFrame>::value, "This needs to be standard layout");
|
||||
static_assert(sizeof(CpuStateFrame::SynchronousFaultData) == 8, "This needs to be 8 bytes");
|
||||
static_assert(std::alignment_of_v<CpuStateFrame::SynchronousFaultDataStruct> == 8, "This needs to be 8 bytes");
|
||||
static_assert(offsetof(CpuStateFrame, SynchronousFaultData) % 8 == 0, "This needs to be aligned");
|
||||
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
|
||||
|
||||
@@ -27,6 +27,7 @@ class HostFeatures final {
|
||||
bool SupportsSHA{};
|
||||
bool SupportsBMI1{};
|
||||
bool SupportsBMI2{};
|
||||
bool SupportsPMULL_128Bit{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
|
||||
@@ -8,8 +8,8 @@
|
||||
#include <FEXCore/Utils/InterruptableConditionVariable.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <map>
|
||||
#include <unordered_map>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -101,7 +101,7 @@ namespace FEXCore::Core {
|
||||
std::unique_ptr<FEXCore::CPU::CPUBackend> CPUBackend;
|
||||
std::unique_ptr<FEXCore::LookupCache> LookupCache;
|
||||
|
||||
std::unordered_map<uint64_t, LocalIREntry> DebugStore;
|
||||
tsl::robin_map<uint64_t, LocalIREntry> DebugStore;
|
||||
|
||||
std::unique_ptr<FEXCore::Frontend::Decoder> FrontendDecoder;
|
||||
std::unique_ptr<FEXCore::IR::PassManager> PassManager;
|
||||
@@ -115,7 +115,7 @@ namespace FEXCore::Core {
|
||||
|
||||
std::shared_mutex ObjectCacheRefCounter{};
|
||||
bool DestroyedByParent{false}; // Should the parent destroy this thread, or it destory itself
|
||||
|
||||
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
|
||||
|
||||
};
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#pragma once
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
|
||||
@@ -280,6 +280,11 @@ public:
|
||||
return Wrapper.GetNode(GetListData());
|
||||
}
|
||||
|
||||
///< Gets an OrderedNode from the IRListView as an OrderedNodeWrapper.
|
||||
[[nodiscard]] OrderedNodeWrapper WrapNode(OrderedNode *Node) const {
|
||||
return Node->Wrapped(GetListData());
|
||||
}
|
||||
|
||||
private:
|
||||
struct BlockRange {
|
||||
using iterator = NodeIterator;
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
#include <time.h>
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
|
||||
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
// clock_gettime will do a VDSO call with the least amount of overhead
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return ts.tv_sec * 1'000'000'000ULL + ts.tv_nsec;
|
||||
}
|
||||
|
||||
// A class that follows scoping rules to generate a profile duration block
|
||||
class ProfilerBlock final {
|
||||
public:
|
||||
ProfilerBlock(std::string_view const Format);
|
||||
|
||||
~ProfilerBlock();
|
||||
|
||||
private:
|
||||
uint64_t DurationBegin;
|
||||
std::string_view const Format;
|
||||
};
|
||||
|
||||
#define UniqueScopeName2(name, line) name ## line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) \
|
||||
FEXCore::Profiler::ProfilerBlock UniqueScopeName(ScopedBlock_, __LINE__) (name)
|
||||
|
||||
#else
|
||||
[[maybe_unused]] static void Init() {}
|
||||
[[maybe_unused]] static void Shutdown() {}
|
||||
[[maybe_unused]] static void TraceObject(std::string_view const Format) {}
|
||||
[[maybe_unused]] static void TraceObject(std::string_view const, uint64_t) {}
|
||||
|
||||
#define FEXCORE_PROFILE_INSTANT(...) do {} while(0)
|
||||
#define FEXCORE_PROFILE_SCOPED(...) do {} while(0)
|
||||
#endif
|
||||
}
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 2b55157592...98f440ce68.
Vendored
+1
-1
Submodule External/vixl updated: 423cd04a70...af65c2974e.
@@ -1,74 +1,64 @@
|
||||
add_library(FEXHeaderUtils INTERFACE)
|
||||
|
||||
# Check for syscall support here
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <sched.h>
|
||||
int main() {
|
||||
return ::getcpu(nullptr, nullptr);
|
||||
}"
|
||||
"
|
||||
#include <sched.h>
|
||||
int main() {
|
||||
return ::getcpu(nullptr, nullptr);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has getcpu helper")
|
||||
add_definitions(-DHAS_SYSCALL_GETCPU=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_GETCPU=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <unistd.h>
|
||||
int main() {
|
||||
return ::gettid();
|
||||
}"
|
||||
"
|
||||
#include <unistd.h>
|
||||
int main() {
|
||||
return ::gettid();
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has gettid helper")
|
||||
add_definitions(-DHAS_SYSCALL_GETTID=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_GETTID=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <signal.h>
|
||||
int main() {
|
||||
return ::tgkill(0, 0, 0);
|
||||
}"
|
||||
"
|
||||
#include <signal.h>
|
||||
int main() {
|
||||
return ::tgkill(0, 0, 0);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has tgkill helper")
|
||||
add_definitions(-DHAS_SYSCALL_TGKILL=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_TGKILL=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <sys/stat.h>
|
||||
int main() {
|
||||
return ::statx(0, nullptr, 0, 0, nullptr);
|
||||
}"
|
||||
"
|
||||
#include <sys/stat.h>
|
||||
int main() {
|
||||
return ::statx(0, nullptr, 0, 0, nullptr);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has statx helper")
|
||||
add_definitions(-DHAS_SYSCALL_STATX=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_STATX=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <stdio.h>
|
||||
int main() {
|
||||
return ::renameat2(0, nullptr, 0, nullptr, 0);
|
||||
}"
|
||||
"
|
||||
#include <stdio.h>
|
||||
int main() {
|
||||
return ::renameat2(0, nullptr, 0, nullptr, 0);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has renameat2 helper")
|
||||
add_definitions(-DHAS_SYSCALL_RENAMEAT2=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <stdio.h>
|
||||
#include <syscall.h>
|
||||
int main() {
|
||||
return ::syscall(SYS_pidfd_open, ::getpid(), 0);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has pidfd_open helper")
|
||||
add_definitions(-DHAS_SYSCALL_PIDFD_OPEN=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_RENAMEAT2=1)
|
||||
endif ()
|
||||
|
||||
target_include_directories(FEXHeaderUtils INTERFACE .)
|
||||
@@ -38,10 +38,15 @@ namespace FHU::Syscalls {
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// Common syscall numbers
|
||||
#ifndef SYS_pidfd_open
|
||||
#define SYS_pidfd_open 434
|
||||
#endif
|
||||
|
||||
inline int32_t getcpu(uint32_t *cpu, uint32_t *node) {
|
||||
// Third argument is unused
|
||||
#if defined(HAS_SYSCALL_GETCPU) && HAS_SYSCALL_GETCPU
|
||||
return ::getcpu(cpu, node, nullptr);
|
||||
return ::getcpu(cpu, node);
|
||||
#else
|
||||
return ::syscall(SYS_getcpu, cpu, node, nullptr);
|
||||
#endif
|
||||
@@ -57,7 +62,7 @@ inline int32_t gettid() {
|
||||
|
||||
inline int32_t tgkill(pid_t tgid, pid_t tid, int sig) {
|
||||
#if defined(HAS_SYSCALL_GETTID) && HAS_SYSCALL_GETTID
|
||||
return ::tgkill(tggid, tid, sig);
|
||||
return ::tgkill(tgid, tid, sig);
|
||||
#else
|
||||
return ::syscall(SYS_tgkill, tgid, tid, sig);
|
||||
#endif
|
||||
@@ -65,7 +70,7 @@ inline int32_t tgkill(pid_t tgid, pid_t tid, int sig) {
|
||||
|
||||
inline int32_t statx(int dirfd, const char *pathname, int32_t flags, uint32_t mask, void *statxbuf) {
|
||||
#if defined(HAS_SYSCALL_STATX) && HAS_SYSCALL_STATX
|
||||
return ::statx(dirfd, pathname, flags, mask, statxbuf);
|
||||
return ::statx(dirfd, pathname, flags, mask, reinterpret_cast<struct statx *__restrict>(statxbuf));
|
||||
#else
|
||||
return ::syscall(SYS_statx, dirfd, pathname, flags, mask, statxbuf);
|
||||
#endif
|
||||
@@ -80,11 +85,7 @@ inline int32_t renameat2(int olddirfd, const char *oldpath, int newdirfd, const
|
||||
}
|
||||
|
||||
inline int32_t pidfd_open(pid_t pid, unsigned int flags) {
|
||||
#if defined(DHAS_SYSCALL_PIDFD_OPEN) && DHAS_SYSCALL_PIDFD_OPEN
|
||||
return ::syscall(SYS_pidfd_open, pid_t pid, unsigned int flags);
|
||||
#else
|
||||
return -1;
|
||||
#endif
|
||||
return ::syscall(SYS_pidfd_open, pid, flags);
|
||||
}
|
||||
|
||||
}
|
||||
Loaded 100 of 284 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user