mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 08:00:21 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a8c1a36c12 | ||
|
|
acfb24f871 | ||
|
|
ecff5aea71 | ||
|
|
74cb225ccb | ||
|
|
d5db2ccf18 | ||
|
|
cebcf50c65 | ||
|
|
3a3c9101c7 | ||
|
|
ec1c7797f4 | ||
|
|
313528c34b | ||
|
|
907fd6b04b | ||
|
|
aa0fc9071e | ||
|
|
0bd924eb7e | ||
|
|
852109c142 | ||
|
|
9b0bb29d78 | ||
|
|
c3e71de1d7 | ||
|
|
0256d6820c | ||
|
|
1b18bfaff5 | ||
|
|
6065e7a62b | ||
|
|
8aecdc536c | ||
|
|
86a2e9e655 | ||
|
|
8f50106187 | ||
|
|
02d3a319f9 | ||
|
|
d18d0435ae | ||
|
|
cdaf1c5262 | ||
|
|
4b0e3bff54 | ||
|
|
c4f7b27459 | ||
|
|
3eb8be9953 | ||
|
|
1f08f8df0d | ||
|
|
0038a0b19c | ||
|
|
166a7c7e53 | ||
|
|
8ac296bd6f | ||
|
|
e092a38e0f | ||
|
|
b145e894e4 | ||
|
|
63a8b66b28 | ||
|
|
a16cc87852 | ||
|
|
1050b60057 | ||
|
|
9188e85164 | ||
|
|
f9b369c550 | ||
|
|
949b205f42 | ||
|
|
31ee8d8178 | ||
|
|
3155590e87 | ||
|
|
feab0bce4b | ||
|
|
0599d80b13 | ||
|
|
d58e12c5e4 | ||
|
|
68939c5a5c | ||
|
|
25c4fb8508 | ||
|
|
88e6c48db7 | ||
|
|
218b0d491a | ||
|
|
bfba74dab9 | ||
|
|
5708846a07 | ||
|
|
fd28783f85 | ||
|
|
0ca34d11ad | ||
|
|
bf3275ba4a | ||
|
|
d8e4e00b2b | ||
|
|
8b38a6dd08 | ||
|
|
63f72621fc | ||
|
|
580a8c9c61 | ||
|
|
249351cef4 | ||
|
|
46690ae352 | ||
|
|
68b5a90518 | ||
|
|
cb54823622 | ||
|
|
3a2ca41724 | ||
|
|
dd4e6d29ca | ||
|
|
497ee32f59 | ||
|
|
d34c287f69 | ||
|
|
b744ca16e1 | ||
|
|
573c262bb2 | ||
|
|
ff2c2f1e1f | ||
|
|
8058391955 | ||
|
|
54dbb9f248 | ||
|
|
7d1351f402 | ||
|
|
21233430aa | ||
|
|
bd1d6820b7 | ||
|
|
18e84ae2e8 | ||
|
|
f9c29056e6 | ||
|
|
40b1c32008 | ||
|
|
b5ed804578 | ||
|
|
fbe3a86c4e | ||
|
|
498c86b47d | ||
|
|
b8b6f81c44 | ||
|
|
aab9e1b751 | ||
|
|
ec976f3f75 | ||
|
|
f169fa5da2 | ||
|
|
35ee12e7e9 | ||
|
|
4497ab8844 | ||
|
|
9f399f3313 | ||
|
|
3939213336 | ||
|
|
cc27a0f666 | ||
|
|
1ff2216063 | ||
|
|
b8165813b4 | ||
|
|
2215b153db | ||
|
|
cb91d585c3 | ||
|
|
c971d4044f | ||
|
|
fdb4a078f2 | ||
|
|
42ea711850 | ||
|
|
96721978c0 | ||
|
|
d64698e4c9 | ||
|
|
4565f2b689 | ||
|
|
5fa7f1d50d | ||
|
|
2cfc42bd6a | ||
|
|
003de2b659 | ||
|
|
4ab6c2252d | ||
|
|
8f9d818368 | ||
|
|
4ebc307744 | ||
|
|
0b519b29d9 | ||
|
|
8881e8d96e | ||
|
|
9a99608f68 | ||
|
|
26c26308db | ||
|
|
123c5d809e | ||
|
|
7c42c7798c | ||
|
|
0f98daf1d9 | ||
|
|
c3de7c63b4 | ||
|
|
ab9123a427 | ||
|
|
e504a8c979 | ||
|
|
f6dd87a3a1 | ||
|
|
89c530054d | ||
|
|
f87edbe1cb | ||
|
|
4aa477de3e | ||
|
|
694e674fe6 | ||
|
|
7ebc0f32b8 | ||
|
|
68cacc2fc3 | ||
|
|
43eb597044 | ||
|
|
7efd827e78 | ||
|
|
2f6f8b93e9 | ||
|
|
d48413cbec | ||
|
|
6bafca688b | ||
|
|
53356f1aa7 | ||
|
|
51281f6a3a | ||
|
|
7a5e08c5ab | ||
|
|
642903a7bf | ||
|
|
f51fd6c78d | ||
|
|
03cf15a9e1 | ||
|
|
fe19c04f6a | ||
|
|
b86cbda03d | ||
|
|
b1af6e23cf | ||
|
|
e26b9b12fa | ||
|
|
9bf47b3f23 | ||
|
|
2887416b2e | ||
|
|
9a92f6f743 | ||
|
|
4929480719 | ||
|
|
c8437d2303 | ||
|
|
ef25ae4d3b | ||
|
|
9399790c11 | ||
|
|
ab7f6484cf | ||
|
|
2d8c5da379 | ||
|
|
b18f148575 | ||
|
|
6426428718 | ||
|
|
af2ee426f5 | ||
|
|
6c4e9ff42d | ||
|
|
62a37a7d70 | ||
|
|
ef83addc74 | ||
|
|
fb308f5947 | ||
|
|
cedb93c11c | ||
|
|
61d2a09827 | ||
|
|
9c6423f37a | ||
|
|
e18a661b50 | ||
|
|
16e5777816 | ||
|
|
64020e8828 | ||
|
|
30798556fc | ||
|
|
cd3518d99d | ||
|
|
bc87c3d494 | ||
|
|
726228c418 | ||
|
|
929b111648 | ||
|
|
1700a73382 | ||
|
|
0f9d791911 | ||
|
|
61acc76be7 | ||
|
|
0f8ba5bf32 | ||
|
|
2cb8a96f0e | ||
|
|
1102122639 | ||
|
|
66e026a9e8 | ||
|
|
b901b42417 | ||
|
|
9f2f10a65f | ||
|
|
b0b41d00ee | ||
|
|
2d56f5eba0 | ||
|
|
8ffc6fbc6b | ||
|
|
0ffd94dadb | ||
|
|
585320093a | ||
|
|
bdd351a42c | ||
|
|
feaee702e9 | ||
|
|
d89fc84fcf | ||
|
|
688cd1a4bb | ||
|
|
c056875a00 | ||
|
|
054118139f | ||
|
|
447148d95a | ||
|
|
f6b1e42a0b | ||
|
|
90e74e5572 | ||
|
|
e4fc5fe5be | ||
|
|
809b2c6115 | ||
|
|
659538ef4b | ||
|
|
4b677f4b42 | ||
|
|
87699ea5a0 | ||
|
|
a2ae113ee9 | ||
|
|
901e2c75d4 | ||
|
|
871d140b7c | ||
|
|
2f6ae1ad02 | ||
|
|
806e98925c | ||
|
|
c7dbd2fac2 | ||
|
|
1b7729efed | ||
|
|
6d1d5aeffb | ||
|
|
9ae04b5771 | ||
|
|
fc61e3b1b5 | ||
|
|
fa1d9910e4 | ||
|
|
d425873eed | ||
|
|
d89c54dd95 | ||
|
|
5e023c55bd | ||
|
|
2487172df3 | ||
|
|
bdd078df17 | ||
|
|
db75335ad0 | ||
|
|
531ab5b5b1 | ||
|
|
0601a863f9 | ||
|
|
10d34c9564 | ||
|
|
a37d6a3841 | ||
|
|
d5234a43da | ||
|
|
b7733540c1 | ||
|
|
2ae02ded74 | ||
|
|
2e57ad644d | ||
|
|
031afbfb18 | ||
|
|
377ce2e2f6 | ||
|
|
a2fd3c077d | ||
|
|
0f5aff73ab | ||
|
|
f0b208e692 | ||
|
|
7fcb5fc590 | ||
|
|
b76f819759 | ||
|
|
4b36d4f1ea | ||
|
|
f4d0c6c807 | ||
|
|
79ab76b42e | ||
|
|
8c3ac61f8d | ||
|
|
a8bc20f76b | ||
|
|
53a55baf23 | ||
|
|
63d6800cf3 | ||
|
|
44492b4828 | ||
|
|
8ed9f8aef5 | ||
|
|
711021e24e | ||
|
|
55dea335fe | ||
|
|
7097532ddf | ||
|
|
66841ce35c | ||
|
|
3d814cb7c1 | ||
|
|
07afdca58b | ||
|
|
93ed346fd2 | ||
|
|
c9eee9bf7f | ||
|
|
d1a4029bc5 | ||
|
|
97070aad25 | ||
|
|
39640185a3 | ||
|
|
4b17506ffe | ||
|
|
2435ebecbe | ||
|
|
410a35b968 | ||
|
|
c8d234a767 | ||
|
|
2bf87ff40f | ||
|
|
ba347c49c9 | ||
|
|
81434cd233 | ||
|
|
46d0df9cba | ||
|
|
b7f58e68c5 | ||
|
|
6ba2accbdc | ||
|
|
29ee94d331 | ||
|
|
cdae654fe4 | ||
|
|
a506c84bc7 | ||
|
|
8e4a47181b | ||
|
|
b5241e0f60 | ||
|
|
57ed466a7f | ||
|
|
4f46f55f2d | ||
|
|
69cfc78ee1 | ||
|
|
04f1ab8571 | ||
|
|
95694b2017 | ||
|
|
00bed2f0c0 | ||
|
|
e718fc35f8 | ||
|
|
717015bae8 | ||
|
|
530d3d809b | ||
|
|
596b32d15c | ||
|
|
8d3918b4f0 | ||
|
|
4c7e31513b | ||
|
|
765509d7f5 | ||
|
|
dbb58d10a6 | ||
|
|
d448976b3c | ||
|
|
de6931b1f5 | ||
|
|
35268d185e | ||
|
|
beef9eee0a | ||
|
|
34a274d4e6 | ||
|
|
ef28a6c19a | ||
|
|
50b5971ee5 | ||
|
|
9a70ae18ea | ||
|
|
d10853b775 | ||
|
|
cdf6a16efc | ||
|
|
02d7261f51 | ||
|
|
9c9ddeffbe | ||
|
|
afa8b3a5c9 | ||
|
|
bbcd4c168c | ||
|
|
d22bd9cac7 | ||
|
|
5ccf25196e | ||
|
|
caf15a2dac | ||
|
|
982a05450c | ||
|
|
3e381b742c | ||
|
|
6b82664166 | ||
|
|
bb30a2eb1e | ||
|
|
b3fdf5c48f | ||
|
|
17d5ed847f | ||
|
|
4335d17fc0 | ||
|
|
ae07958577 | ||
|
|
c51b9ba3d6 | ||
|
|
cc6ff5e9e6 | ||
|
|
b968ea7e7e | ||
|
|
d14b6e160e | ||
|
|
b09b9488ef | ||
|
|
a8120ee7ef | ||
|
|
fb82059750 | ||
|
|
0fc6240d72 | ||
|
|
a7c6fdb1fc | ||
|
|
b76a2963cf | ||
|
|
42b0fbd34c | ||
|
|
f69ef8606f | ||
|
|
afa5ad5f9f | ||
|
|
1fc82708e9 | ||
|
|
917cbbadde | ||
|
|
c37dc81839 | ||
|
|
df718d55ef | ||
|
|
54412f1d5e | ||
|
|
da76023bea | ||
|
|
41e9309a36 | ||
|
|
116268b275 | ||
|
|
73802492b8 | ||
|
|
7cd4d53fa9 | ||
|
|
c44757975e | ||
|
|
f25cdcdf63 | ||
|
|
5d37253e85 | ||
|
|
cd5f42ec79 | ||
|
|
6651f9e94b | ||
|
|
e3ee579f92 | ||
|
|
73e7240574 | ||
|
|
391f9aa97d | ||
|
|
8ad54e7bd5 | ||
|
|
9eccc01dd3 | ||
|
|
39c1f816fc | ||
|
|
a32b892787 | ||
|
|
1b144ba3f0 | ||
|
|
6a39a8db72 | ||
|
|
602c530615 | ||
|
|
c8c27f26f7 | ||
|
|
549cdc4c2c | ||
|
|
3160e0a430 | ||
|
|
dcebe85f3a | ||
|
|
2ba0b66426 | ||
|
|
5c9543f159 | ||
|
|
f6e3689f30 | ||
|
|
a761343717 | ||
|
|
b4c47a3d24 | ||
|
|
906988c49b | ||
|
|
4186b2ad82 | ||
|
|
2943cff73f | ||
|
|
53ac5579fb | ||
|
|
0d7a9f911a | ||
|
|
dce9de222d | ||
|
|
d46722a95c | ||
|
|
3ba4da7736 | ||
|
|
6abf5b90b7 | ||
|
|
00aa4ddea0 | ||
|
|
d0c6f9de22 | ||
|
|
a85cc85081 | ||
|
|
75793300f2 | ||
|
|
672805584e | ||
|
|
5b4fd590d1 | ||
|
|
0ccd38f593 | ||
|
|
b46e5d4488 | ||
|
|
d7223d598f | ||
|
|
7a0368132d | ||
|
|
aff3914a66 | ||
|
|
8876047875 | ||
|
|
78e2aa16f0 | ||
|
|
3ef695cf70 | ||
|
|
c4d8dd6413 | ||
|
|
a49d30f6e2 | ||
|
|
512643d3d6 | ||
|
|
b3a69af752 | ||
|
|
64c0dc47a9 | ||
|
|
923c323d6f | ||
|
|
ee47b5bbc9 | ||
|
|
e8cd655c84 | ||
|
|
d80daf2692 | ||
|
|
0e5c9e8b06 | ||
|
|
6bc4aed82c | ||
|
|
e69e1200f5 | ||
|
|
81e253b06b | ||
|
|
9af52fb642 | ||
|
|
eaddd44d17 | ||
|
|
854e699589 | ||
|
|
43dcc84c07 | ||
|
|
6e9d5f00de | ||
|
|
a65ca9663f | ||
|
|
02b767c0ea | ||
|
|
4b1c1d266d | ||
|
|
d9bf140971 | ||
|
|
1402776ba6 | ||
|
|
e36fb47d98 | ||
|
|
2bb37357c0 | ||
|
|
7494ac7615 | ||
|
|
44bc3fb90b | ||
|
|
20b00ecc9b | ||
|
|
40662f947f | ||
|
|
6e01934edc | ||
|
|
151fc5e97f | ||
|
|
5f431dc776 | ||
|
|
5ec4d3125a | ||
|
|
8e511e7db4 | ||
|
|
99b8046f03 | ||
|
|
d39dea1ae3 | ||
|
|
e94643d5ca | ||
|
|
1becbab0dc | ||
|
|
d49efb451e | ||
|
|
d2a56ebd8c | ||
|
|
0e11a9b7ac | ||
|
|
68c77d12d4 | ||
|
|
423e29ba42 | ||
|
|
dcfbc2f20e | ||
|
|
0b29c99fed | ||
|
|
18556a9f75 | ||
|
|
02d93782ba | ||
|
|
62cfc26262 | ||
|
|
b01a6b94e7 | ||
|
|
713ebf1476 | ||
|
|
26e50efdb2 | ||
|
|
c8928999bf | ||
|
|
e1f378c6cf | ||
|
|
d10222329c | ||
|
|
176fa7ab1d | ||
|
|
ebb7137839 | ||
|
|
d7092a1231 | ||
|
|
55cbb0b340 | ||
|
|
db7fb56e9d | ||
|
|
2bed7440a8 | ||
|
|
0019bdecef | ||
|
|
51d355da30 | ||
|
|
28170fd723 | ||
|
|
44c65c35c8 | ||
|
|
d8f8daf48a | ||
|
|
b148cc6ca3 | ||
|
|
2a4c169fff | ||
|
|
c2c84e4bd8 | ||
|
|
c2f8b5b1ba | ||
|
|
3a33f554a0 | ||
|
|
11ce97655b | ||
|
|
dc866538d4 | ||
|
|
48ad9e9a87 | ||
|
|
c75778abeb | ||
|
|
fb2a59a67f | ||
|
|
f4c92756fc | ||
|
|
657c27556c | ||
|
|
42e68d8544 | ||
|
|
ae69c4d895 | ||
|
|
5a0db4d812 | ||
|
|
ddd241fe39 | ||
|
|
3dc7b8d90a | ||
|
|
0bccb1ece5 | ||
|
|
bf1e319d90 | ||
|
|
bc6ae7feb4 | ||
|
|
8d6a43d708 | ||
|
|
bd1bca2c3a | ||
|
|
9858ab7388 | ||
|
|
1f6b69573c | ||
|
|
1c8c5b77f1 | ||
|
|
e9bd037cf9 | ||
|
|
6f8353ab28 | ||
|
|
c25720429d | ||
|
|
276e9aded3 | ||
|
|
b88ac3359d | ||
|
|
264f3be8b4 | ||
|
|
4282f96d35 | ||
|
|
9cdd759fc1 | ||
|
|
403e8f8702 | ||
|
|
2e989e4262 | ||
|
|
5666a352d4 | ||
|
|
1aa8c6f996 | ||
|
|
9882f53613 | ||
|
|
8760c593ec | ||
|
|
ad695bdd59 | ||
|
|
adff4bb1d7 | ||
|
|
42c931cf22 | ||
|
|
5d44dea47c | ||
|
|
56c95e3b36 | ||
|
|
840f306a7d | ||
|
|
9def89d5f8 | ||
|
|
84c2f93dab | ||
|
|
b30733e2a7 | ||
|
|
da58e6a597 | ||
|
|
229e7c5b61 | ||
|
|
26685143be | ||
|
|
e54b9237c6 | ||
|
|
ac1b6d9482 | ||
|
|
bb6e98a6fc | ||
|
|
32c75f06b3 | ||
|
|
a5de2d1008 | ||
|
|
f841912c75 | ||
|
|
3b8c36882d | ||
|
|
fca4c7e6bf | ||
|
|
5ffc611d13 | ||
|
|
130f02647b | ||
|
|
981eea6ade | ||
|
|
3f788eb4a8 | ||
|
|
486dc974c4 | ||
|
|
3b1fbbc766 | ||
|
|
d01db8f293 | ||
|
|
fd09ded049 | ||
|
|
8e2b4a306d | ||
|
|
1d58f38aa5 | ||
|
|
5ff9a83b07 | ||
|
|
f5decb5f83 | ||
|
|
11fc49a0f8 | ||
|
|
48c03d747a | ||
|
|
643750817a | ||
|
|
a52dd71e44 | ||
|
|
8c02bd43df | ||
|
|
f635a12129 | ||
|
|
8191c4905b | ||
|
|
58a034b79d | ||
|
|
2d53867668 | ||
|
|
8a57fc5838 | ||
|
|
5481e6d79a | ||
|
|
c852a58ee3 | ||
|
|
8cfc016b3f | ||
|
|
8c94b782c6 | ||
|
|
1dce4919f2 | ||
|
|
cbda688e29 | ||
|
|
159ed07e68 | ||
|
|
a18b2d0e17 | ||
|
|
4c9adab58d | ||
|
|
4c9f1b105d | ||
|
|
2290353295 | ||
|
|
c16bf09310 | ||
|
|
90db9486ce | ||
|
|
b79faa6207 | ||
|
|
2293d3067a | ||
|
|
de431f113e | ||
|
|
34e265a801 | ||
|
|
a668492fb7 | ||
|
|
da069571f3 | ||
|
|
d2bac45b49 | ||
|
|
8913c59acc | ||
|
|
c3261b4aeb | ||
|
|
c7fb95aec5 | ||
|
|
c00cef6dc1 | ||
|
|
429ff94dc5 | ||
|
|
a6c67ca749 | ||
|
|
f51812a670 | ||
|
|
686294f1c4 | ||
|
|
a47ed105e7 | ||
|
|
b2d579a268 | ||
|
|
eb1050092f | ||
|
|
b3794f5541 | ||
|
|
1ecfa3253d | ||
|
|
5daf007b6a | ||
|
|
8efa5febd0 | ||
|
|
5fee8028cd | ||
|
|
6bc7a83c64 | ||
|
|
e55b5d0d11 | ||
|
|
19de7f2785 | ||
|
|
e32c5384ab | ||
|
|
b391fe6b92 | ||
|
|
4cfb81156f | ||
|
|
6121708e55 | ||
|
|
6ab214adea | ||
|
|
90b1ac4162 | ||
|
|
3abe6c14a1 | ||
|
|
fc1b500eff | ||
|
|
5d47b9195b | ||
|
|
a8272b74f6 | ||
|
|
12dc16780f | ||
|
|
b8af569841 | ||
|
|
8bee101795 | ||
|
|
2d66bc258a | ||
|
|
d2f86e49f7 | ||
|
|
d66cd16cfb | ||
|
|
04e785e434 | ||
|
|
15a1a0f7d9 | ||
|
|
0a58ce6134 | ||
|
|
8f5607f0e8 | ||
|
|
a21789d3d8 | ||
|
|
efd6e95059 | ||
|
|
9bdb1f4306 | ||
|
|
ae4b7135d5 | ||
|
|
d503366816 | ||
|
|
0fe2827fcc | ||
|
|
cd6722f77b | ||
|
|
ffb745b662 | ||
|
|
bb10f25808 | ||
|
|
aa1076d12b | ||
|
|
1e827ec7a6 | ||
|
|
3fe2650787 | ||
|
|
3e99e814bc | ||
|
|
2019f8138e | ||
|
|
e44d1f136b | ||
|
|
09872402df | ||
|
|
b078a41a02 | ||
|
|
3a5eeb5700 | ||
|
|
9433ae3405 | ||
|
|
4658b24f9a | ||
|
|
4e7d0e6be0 | ||
|
|
4ddd98708f | ||
|
|
c161fd218c | ||
|
|
7e257cc268 | ||
|
|
d8ef70280c | ||
|
|
ec003281be | ||
|
|
e58f67b76c | ||
|
|
57178abcd2 | ||
|
|
73ca4f8314 | ||
|
|
527752c25b | ||
|
|
38fa866c91 | ||
|
|
c902b8807a | ||
|
|
4934c1fd94 | ||
|
|
766fbe3db3 | ||
|
|
29405f2690 | ||
|
|
51f505acca | ||
|
|
77415538f7 | ||
|
|
9fb69ed206 | ||
|
|
735a4f90db | ||
|
|
7ef8dc13ba | ||
|
|
656477ec63 | ||
|
|
d080180e85 | ||
|
|
af1d2d6005 | ||
|
|
90c1282f3a | ||
|
|
d5d7eec8b0 | ||
|
|
5337b9537d | ||
|
|
e72c016230 | ||
|
|
072cf4c5bd | ||
|
|
27ededf47f | ||
|
|
82d7f9fdd7 | ||
|
|
d85153d6b3 | ||
|
|
6b698e6cd1 | ||
|
|
9475f79ec6 | ||
|
|
f906c6a0f4 | ||
|
|
e88c92de57 | ||
|
|
48ed906a7b | ||
|
|
b03b02d2f2 | ||
|
|
d00d476a0a | ||
|
|
ac1e32994a | ||
|
|
800d447f3d | ||
|
|
8111b7cc7f | ||
|
|
a86c922073 | ||
|
|
46fb8583bb | ||
|
|
3487d120ec | ||
|
|
07394d6a6e | ||
|
|
8d3204171c | ||
|
|
7641f722e9 | ||
|
|
b51fa497c5 | ||
|
|
34722bed3d | ||
|
|
981c3009ee | ||
|
|
38cf357d85 | ||
|
|
2533ed4a63 | ||
|
|
5a4691fdfc | ||
|
|
6c035a0d61 | ||
|
|
a234aa300d | ||
|
|
f6abbedbd1 | ||
|
|
b6fe4cd6dd | ||
|
|
bdae4f6915 | ||
|
|
f8b6edfb2b | ||
|
|
0a1ecdf6ae | ||
|
|
beec203f56 | ||
|
|
d323032ec9 | ||
|
|
7472b21f33 | ||
|
|
1058575d3a | ||
|
|
e9867ca35a | ||
|
|
84277319fa | ||
|
|
8f8aa55c7f | ||
|
|
72a4063651 | ||
|
|
0b1229da55 | ||
|
|
1d3ce30e50 | ||
|
|
71187d3ad7 | ||
|
|
dd8a3a9aea | ||
|
|
7b2fc37651 | ||
|
|
426569d74d | ||
|
|
e877d5b82c | ||
|
|
572e0d04d5 | ||
|
|
e3d7161ac5 | ||
|
|
fcbf0de05a |
No files matched your search
@@ -3,10 +3,13 @@
|
||||
# Ignore all files in the External directory
|
||||
External/*
|
||||
|
||||
# SoftFloat-3e code doesn't belong to us
|
||||
# SoftFloat-3e code doesn't belong to us
|
||||
FEXCore/Source/Common/SoftFloat-3e/*
|
||||
Source/Common/cpp-optparse/*
|
||||
|
||||
# Files with human-indented tables for readability - don't mess with these
|
||||
FEXCore/Source/Interface/Core/X86Tables/*
|
||||
|
||||
# Inline headers with list-like content that can't be processed individually
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
|
||||
@@ -13,3 +13,6 @@
|
||||
|
||||
# Second reformat to find fixed point PR#3577
|
||||
905aa935f5ce344a48ef4d5edab3c31efa8d793e
|
||||
|
||||
# Reformat of CodeEmitter inl files
|
||||
8760c593ece92d7e9fa94c40da0368fd367c9cad
|
||||
@@ -250,7 +250,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -184,7 +184,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -97,7 +97,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -128,7 +128,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
@@ -137,7 +137,7 @@ jobs:
|
||||
|
||||
- name: Upload results InstCountCI
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-instcountci
|
||||
|
||||
@@ -92,7 +92,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -28,7 +28,7 @@ jobs:
|
||||
|
||||
- name: Get changed files
|
||||
id: changed-files
|
||||
uses: tj-actions/changed-files@v39
|
||||
uses: step-security/changed-files@3dbe17c78367e7d60f00d78ae6781a35be47b4a1 # v45.0.1
|
||||
with:
|
||||
separator: ","
|
||||
skip_initial_fetch: true
|
||||
|
||||
@@ -126,7 +126,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
+3
-4
@@ -5,10 +5,6 @@
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
url = https://github.com/herumi/xbyak.git
|
||||
[submodule "External/fex-posixtest-bins"]
|
||||
shallow = true
|
||||
path = External/fex-posixtest-bins
|
||||
@@ -47,3 +43,6 @@
|
||||
[submodule "External/jemalloc_glibc"]
|
||||
path = External/jemalloc_glibc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
[submodule "External/tracy"]
|
||||
path = External/tracy
|
||||
url = https://github.com/wolfpld/tracy
|
||||
+28
-15
@@ -8,7 +8,7 @@ option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" TRUE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
@@ -26,17 +26,17 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only useful for CI testing)" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
@@ -61,6 +61,22 @@ if (ENABLE_FEXCORE_PROFILER)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
elseif (FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=2)
|
||||
add_definitions(-DTRACY_ENABLE=1)
|
||||
# Required so that Tracy will only start in the selected guest application
|
||||
add_definitions(-DTRACY_MANUAL_LIFETIME=1)
|
||||
add_definitions(-DTRACY_DELAYED_INIT=1)
|
||||
# This interferes with FEX's signal handling
|
||||
add_definitions(-DTRACY_NO_CRASH_HANDLER=1)
|
||||
# Tracy can gather call stack samples in regular intervals, but this
|
||||
# isn't useful for us since it would usually sample opaque JIT code
|
||||
add_definitions(-DTRACY_NO_SAMPLING=1)
|
||||
# This pulls in libbacktrace which allocators in global constructors (before FEX can set up its allocator hooks)
|
||||
add_definitions(-DTRACY_NO_CALLSTACK=1)
|
||||
if (MINGW_BUILD)
|
||||
message(FATAL_ERROR "Tracy profiler not supported")
|
||||
endif()
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
@@ -265,16 +281,15 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
# Enable vixl disassembler if tests are enabled.
|
||||
set(COMPILE_VIXL_DISASSEMBLER TRUE)
|
||||
endif()
|
||||
|
||||
if (COMPILE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_subdirectory(External/tracy)
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
|
||||
@@ -298,7 +313,7 @@ add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
find_package(Catch2 QUIET)
|
||||
find_package(Catch2 3 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
@@ -369,10 +384,6 @@ if (TUNE_CPU STREQUAL "native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
@@ -479,6 +490,7 @@ if (BUILD_THUNKS)
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
@@ -497,6 +509,7 @@ if (BUILD_THUNKS)
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
|
||||
+157
-183
@@ -11,6 +11,14 @@
|
||||
* FEX-Emu ALU operations usually have a 32-bit or 64-bit operating size encoded in the IR operation,
|
||||
* This allows FEX to use a single helper function which decodes to both handlers.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
private:
|
||||
static bool IsADRRange(int64_t Imm) {
|
||||
return Imm >= -1048576 && Imm <= 1048575;
|
||||
@@ -28,26 +36,23 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void adr(ARMEmitter::Register rd, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADR });
|
||||
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adr(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
adr(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
@@ -57,39 +62,34 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void adrp(ARMEmitter::Register rd, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADRP });
|
||||
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adrp(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
adrp(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
adr(rd, Label);
|
||||
}
|
||||
else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL)
|
||||
- (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
// If the range is in the ADRP range then we can use ADRP.
|
||||
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
|
||||
@@ -102,24 +102,22 @@ public:
|
||||
// Now even an add
|
||||
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
Label->Insts.emplace_back(SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN });
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
LongAddressGen(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
LongAddressGen(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
@@ -176,11 +174,7 @@ public:
|
||||
// Logical immediate
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
and_(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -191,11 +185,7 @@ public:
|
||||
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
ands(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -206,22 +196,14 @@ public:
|
||||
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
orr(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -355,8 +337,8 @@ public:
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits,
|
||||
lsb, width);
|
||||
|
||||
bfm(s, rd, rn, lsb, lsb_p_width - 1);
|
||||
}
|
||||
@@ -375,188 +357,142 @@ public:
|
||||
|
||||
// Data processing - 2 source
|
||||
void udiv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0000'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void sdiv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0000'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
|
||||
void lslv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void lsrv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void asrv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void rorv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void crc32b(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32h(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32w(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cb(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32ch(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cw(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0000'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void irg(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0001'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0001'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void gmi(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0001'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0001'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void pacga(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0011'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0011'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32x(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'11U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cx(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'11U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void subps(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b011'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
constexpr uint32_t Op = (0b011'1010'110U << 21) | (0b0000'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Data processing - 1 source
|
||||
void rbit(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'00U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev16(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'01U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'01U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'10U << 10);
|
||||
DataProcessing_1Source(Op, ARMEmitter::Size::i32Bit, rd, rn);
|
||||
}
|
||||
void rev32(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'10U << 10);
|
||||
DataProcessing_1Source(Op, ARMEmitter::Size::i64Bit, rd, rn);
|
||||
}
|
||||
void clz(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'00U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cls(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'01U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'01U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'11U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'11U << 10);
|
||||
DataProcessing_1Source(Op, ARMEmitter::Size::i64Bit, rd, rn);
|
||||
}
|
||||
void rev(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10) |
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'10U << 10) | (s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void ctz(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
@@ -573,27 +509,33 @@ public:
|
||||
orr(ARMEmitter::Size::i32Bit, rd.R(), ARMEmitter::Reg::zr, rn.R(), ARMEmitter::ShiftType::LSL, 0);
|
||||
}
|
||||
|
||||
void mvn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void mvn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
orn(s, rd, ARMEmitter::Reg::zr, rn, Shift, amt);
|
||||
}
|
||||
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b000'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b110'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void bic(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void bic(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b000'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void bics(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void bics(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b110'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
@@ -601,30 +543,36 @@ public:
|
||||
ands(s, Reg::zr, rn, rm, shift, amt);
|
||||
}
|
||||
|
||||
void orn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void orn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b100'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void eon(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void eon(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b100'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
// AddSub - shifted register
|
||||
void add(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void add(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
add(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void adds(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void adds(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void sub(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void neg(ARMEmitter::XRegister rd, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
@@ -633,23 +581,27 @@ public:
|
||||
void cmp(ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void subs(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void subs(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void negs(ARMEmitter::XRegister rd, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(rd, ARMEmitter::XReg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void add(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void add(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
add(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void adds(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void adds(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, ARMEmitter::WReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void sub(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void neg(ARMEmitter::WRegister rd, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
@@ -658,65 +610,78 @@ public:
|
||||
void cmp(ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void subs(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void subs(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void negs(ARMEmitter::WRegister rd, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(rd, ARMEmitter::WReg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b000'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b010'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void cmn(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void cmn(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
adds(s, ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b100'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void neg(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void neg(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
sub(s, rd, ARMEmitter::Reg::zr, rm, Shift, amt);
|
||||
}
|
||||
void cmp(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void cmp(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
subs(s, ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b110'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void negs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void negs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
subs(s, rd, ARMEmitter::Reg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
// AddSub - extended register
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift <= 4, "Shift amount is too large");
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift <= 4, "Shift amount is too large");
|
||||
constexpr uint32_t Op = 0b000'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b010'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void cmn(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
adds(s, ARMEmitter::Reg::zr, rn, rm, Option, Shift);
|
||||
}
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b100'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b110'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
@@ -751,8 +716,8 @@ public:
|
||||
|
||||
// Rotate right into flags
|
||||
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
|
||||
LOGMAN_THROW_AA_FMT(shift <= 63, "Shift must be within 0-63. Shift: {}", shift);
|
||||
LOGMAN_THROW_AA_FMT(mask <= 15, "Mask must be within 0-15. Mask: {}", mask);
|
||||
LOGMAN_THROW_A_FMT(shift <= 63, "Shift must be within 0-63. Shift: {}", shift);
|
||||
LOGMAN_THROW_A_FMT(mask <= 15, "Mask must be within 0-15. Mask: {}", mask);
|
||||
|
||||
uint32_t Op = 0b1011'1010'0000'0000'0000'0100'0000'0000;
|
||||
Op |= rn.Idx() << 5;
|
||||
@@ -816,7 +781,8 @@ public:
|
||||
}
|
||||
void cset(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 0, 0b01, s, rd, ARMEmitter::Reg::zr, ARMEmitter::Reg::zr, static_cast<ARMEmitter::Condition>(FEXCore::ToUnderlying(Cond) ^ FEXCore::ToUnderlying(ARMEmitter::Condition::CC_NE)));
|
||||
ConditionalCompare(Op, 0, 0b01, s, rd, ARMEmitter::Reg::zr, ARMEmitter::Reg::zr,
|
||||
static_cast<ARMEmitter::Condition>(FEXCore::ToUnderlying(Cond) ^ FEXCore::ToUnderlying(ARMEmitter::Condition::CC_NE)));
|
||||
}
|
||||
void csinc(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
@@ -898,8 +864,7 @@ public:
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_AA_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV,
|
||||
"Cannot invert CC_AL or CC_NV");
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
@@ -950,7 +915,7 @@ private:
|
||||
LSL12 = true;
|
||||
Imm >>= 12;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(TooLarge == false, "Imm amount too large: 0x{:x}", Imm);
|
||||
LOGMAN_THROW_A_FMT(TooLarge == false, "Imm amount too large: 0x{:x}", Imm);
|
||||
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
@@ -995,7 +960,8 @@ private:
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void DataProcessing_Logical_Imm(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
void DataProcessing_Logical_Imm(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n,
|
||||
uint32_t immr, uint32_t imms) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1014,9 +980,8 @@ private:
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_AA_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
|
||||
const auto immr = (reg_size_bits - lsb) & (reg_size_bits - 1);
|
||||
const auto imms = width - 1;
|
||||
@@ -1028,12 +993,13 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
void DataProcessing_Extract(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, uint32_t Imm) {
|
||||
void DataProcessing_Extract(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
// Current ARMv8 spec hardcodes SF == N for this class of instructions.
|
||||
// Anythign else is undefined behaviour.
|
||||
const uint32_t N = s == ARMEmitter::Size::i64Bit ? (1U << 22) : 0;
|
||||
const uint32_t N = s == ARMEmitter::Size::i64Bit ? (1U << 22) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -1076,10 +1042,11 @@ private:
|
||||
}
|
||||
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn,
|
||||
ARMEmitter::Register rm, ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_A_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
LOGMAN_THROW_A_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -1097,7 +1064,8 @@ private:
|
||||
}
|
||||
|
||||
// AddSub - extended register
|
||||
void DataProcessing_Extended_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift) {
|
||||
void DataProcessing_Extended_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn,
|
||||
ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1113,7 +1081,8 @@ private:
|
||||
}
|
||||
// Conditional compare - register
|
||||
template<typename T>
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, uint32_t o3, ARMEmitter::Size s, ARMEmitter::Register rn, T rm, ARMEmitter::StatusFlags flags, ARMEmitter::Condition Cond) {
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, uint32_t o3, ARMEmitter::Size s, ARMEmitter::Register rn, T rm,
|
||||
ARMEmitter::StatusFlags flags, ARMEmitter::Condition Cond) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1131,7 +1100,8 @@ private:
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, T rm, ARMEmitter::Condition Cond) {
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, T rm,
|
||||
ARMEmitter::Condition Cond) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1148,7 +1118,8 @@ private:
|
||||
}
|
||||
|
||||
// Data-processing - 3 source
|
||||
void DataProcessing_3Source(uint32_t Op, uint32_t Op0, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::Register ra) {
|
||||
void DataProcessing_3Source(uint32_t Op, uint32_t Op0, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn,
|
||||
ARMEmitter::Register rm, ARMEmitter::Register ra) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1170,4 +1141,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
+882
-1013
File diff suppressed because it is too large.
Load diff
@@ -3,339 +3,325 @@
|
||||
*
|
||||
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
public:
|
||||
// Branches, Exception Generating and System instructions
|
||||
public:
|
||||
// Conditional branch immediate
|
||||
///< Branch conditional
|
||||
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
public:
|
||||
// Conditional branch immediate
|
||||
///< Branch conditional
|
||||
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
} else {
|
||||
b(Cond, &Label->Forward);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
///< Branch consistent conditional
|
||||
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
} else {
|
||||
bc(Cond, &Label->Forward);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void br(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 | 0b0'000 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void blr(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 | 0b0'001 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 | 0b0'010 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
|
||||
// Unconditional branch immediate
|
||||
void b(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void b(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
} else {
|
||||
b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
b(Cond, &Label->Forward);
|
||||
}
|
||||
void bl(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void bl(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
} else {
|
||||
bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
///< Branch consistent conditional
|
||||
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
// Compare and branch
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bc(ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
// Test and branch immediate
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
bc(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
// Unconditional branch register
|
||||
void br(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'000 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void blr(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'001 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'010 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
// Unconditional branch immediate
|
||||
void b(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
}
|
||||
else {
|
||||
b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void bl(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bl(LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
}
|
||||
else {
|
||||
bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Test and branch immediate
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Conditional branch immediate
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
// Conditional branch immediate
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Op1 << 24;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Op0 << 4;
|
||||
Instr |= FEXCore::ToUnderlying(Cond);
|
||||
Instr |= Op1 << 24;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Op0 << 4;
|
||||
Instr |= FEXCore::ToUnderlying(Cond);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Encode_rn(rn);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Unconditional branch register
|
||||
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Encode_rn(rn);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Unconditional branch - immediate
|
||||
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Imm & 0x3FF'FFFF;
|
||||
dc32(Instr);
|
||||
}
|
||||
// Unconditional branch - immediate
|
||||
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Imm & 0x3FF'FFFF;
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
// Compare and branch
|
||||
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
Instr |= SF;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Test and branch - immediate
|
||||
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
// Test and branch - immediate
|
||||
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= (Bit >> 5) << 31;
|
||||
Instr |= (Bit & 0b1'1111) << 19;
|
||||
Instr |= (Imm & 0x3FFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
Instr |= (Bit >> 5) << 31;
|
||||
Instr |= (Bit & 0b1'1111) << 19;
|
||||
Instr |= (Imm & 0x3FFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <type_traits>
|
||||
|
||||
namespace ARMEmitter {
|
||||
class Buffer {
|
||||
@@ -21,29 +22,25 @@ public:
|
||||
Size = BaseSize;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_trivially_copyable_v<T>)
|
||||
void dcn(const T& Data) {
|
||||
std::memcpy(CurrentOffset, &Data, sizeof(Data));
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void dc8(uint8_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc16(uint16_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc32(uint32_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
void dc64(uint64_t Data) {
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc64(uint64_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void EmitString(const char* String) {
|
||||
const auto StringLength = strlen(String);
|
||||
memcpy(CurrentOffset, String, StringLength);
|
||||
@@ -95,10 +92,6 @@ public:
|
||||
|
||||
protected:
|
||||
|
||||
void ResetBuffer() {
|
||||
CurrentOffset = BufferBase;
|
||||
}
|
||||
|
||||
uint8_t* BufferBase;
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
|
||||
@@ -341,94 +341,88 @@ public:
|
||||
};
|
||||
|
||||
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenSystemReg() {
|
||||
return op0 << 19 | op1 << 16 | CRn << 12 | CRm << 8 | op2 << 5;
|
||||
};
|
||||
inline constexpr uint32_t GenSystemReg = op0 << 19 | op1 << 16 | CRn << 12 | CRm << 8 | op2 << 5;
|
||||
|
||||
// This `SystemRegister` enum is used for the mrs/msr instructions.
|
||||
enum class SystemRegister : uint32_t {
|
||||
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
|
||||
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
|
||||
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
|
||||
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
|
||||
TPIDRRO_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b011>(),
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
|
||||
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>,
|
||||
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>,
|
||||
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>,
|
||||
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>,
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>,
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>,
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>,
|
||||
TPIDRRO_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b011>,
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>,
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>,
|
||||
};
|
||||
|
||||
template<uint32_t op1, uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenDCReg() {
|
||||
return op1 << 16 | CRm << 8 | op2 << 5;
|
||||
};
|
||||
inline constexpr uint32_t GenDCReg = op1 << 16 | CRm << 8 | op2 << 5;
|
||||
|
||||
// This `DataCacheOperation` enum is used for the dc instruction.
|
||||
enum class DataCacheOperation : uint32_t {
|
||||
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
|
||||
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
|
||||
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
|
||||
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
|
||||
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
|
||||
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
|
||||
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
|
||||
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
|
||||
IVAC = GenDCReg<0b000, 0b0110, 0b001>,
|
||||
ISW = GenDCReg<0b000, 0b0110, 0b010>,
|
||||
CSW = GenDCReg<0b000, 0b1010, 0b010>,
|
||||
CISW = GenDCReg<0b000, 0b1110, 0b010>,
|
||||
ZVA = GenDCReg<0b011, 0b0100, 0b001>,
|
||||
CVAC = GenDCReg<0b011, 0b1010, 0b001>,
|
||||
CVAU = GenDCReg<0b011, 0b1011, 0b001>,
|
||||
CIVAC = GenDCReg<0b011, 0b1110, 0b001>,
|
||||
|
||||
// MTE2
|
||||
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
|
||||
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
|
||||
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
|
||||
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
|
||||
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
|
||||
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
|
||||
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
|
||||
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
|
||||
IGVAC = GenDCReg<0b000, 0b0110, 0b011>,
|
||||
IGSW = GenDCReg<0b000, 0b0110, 0b100>,
|
||||
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>,
|
||||
IGDSW = GenDCReg<0b000, 0b0110, 0b110>,
|
||||
CGSW = GenDCReg<0b000, 0b1010, 0b100>,
|
||||
CGDSW = GenDCReg<0b000, 0b1010, 0b110>,
|
||||
CIGSW = GenDCReg<0b000, 0b1110, 0b100>,
|
||||
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>,
|
||||
|
||||
// MTE
|
||||
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
|
||||
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
|
||||
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
|
||||
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
|
||||
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
|
||||
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
|
||||
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
|
||||
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
|
||||
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
|
||||
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
|
||||
GVA = GenDCReg<0b011, 0b0100, 0b011>,
|
||||
GZVA = GenDCReg<0b011, 0b0100, 0b100>,
|
||||
CGVAC = GenDCReg<0b011, 0b1010, 0b011>,
|
||||
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>,
|
||||
CGVAP = GenDCReg<0b011, 0b1100, 0b011>,
|
||||
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>,
|
||||
CGVADP = GenDCReg<0b011, 0b1101, 0b011>,
|
||||
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>,
|
||||
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>,
|
||||
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>,
|
||||
|
||||
// DPB
|
||||
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
|
||||
CVAP = GenDCReg<0b011, 0b1100, 0b001>,
|
||||
|
||||
// DPB2
|
||||
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
|
||||
CVADP = GenDCReg<0b011, 0b1101, 0b001>,
|
||||
};
|
||||
|
||||
template<uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenHintBarrierReg() {
|
||||
return CRm << 8 | op2 << 5;
|
||||
}
|
||||
inline constexpr uint32_t GenHintBarrierReg = CRm << 8 | op2 << 5;
|
||||
|
||||
// This `HintRegister` enum is used for the hint instruction.
|
||||
enum class HintRegister : uint32_t {
|
||||
NOP = GenHintBarrierReg<0b0000, 0b000>(),
|
||||
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
|
||||
WFE = GenHintBarrierReg<0b0000, 0b010>(),
|
||||
WFI = GenHintBarrierReg<0b0000, 0b011>(),
|
||||
SEV = GenHintBarrierReg<0b0000, 0b100>(),
|
||||
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
|
||||
DGH = GenHintBarrierReg<0b0000, 0b110>(),
|
||||
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
|
||||
NOP = GenHintBarrierReg<0b0000, 0b000>,
|
||||
YIELD = GenHintBarrierReg<0b0000, 0b001>,
|
||||
WFE = GenHintBarrierReg<0b0000, 0b010>,
|
||||
WFI = GenHintBarrierReg<0b0000, 0b011>,
|
||||
SEV = GenHintBarrierReg<0b0000, 0b100>,
|
||||
SEVL = GenHintBarrierReg<0b0000, 0b101>,
|
||||
DGH = GenHintBarrierReg<0b0000, 0b110>,
|
||||
CSDB = GenHintBarrierReg<0b0010, 0b100>,
|
||||
};
|
||||
|
||||
// This `BarrierRegister` enum is used for the various barrier instructions.
|
||||
enum class BarrierRegister : uint32_t {
|
||||
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
|
||||
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
|
||||
DSB = GenHintBarrierReg<0b0000, 0b100>(),
|
||||
DMB = GenHintBarrierReg<0b0000, 0b101>(),
|
||||
ISB = GenHintBarrierReg<0b0000, 0b110>(),
|
||||
SB = GenHintBarrierReg<0b0000, 0b111>(),
|
||||
CLREX = GenHintBarrierReg<0b0000, 0b010>,
|
||||
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>,
|
||||
DSB = GenHintBarrierReg<0b0000, 0b100>,
|
||||
DMB = GenHintBarrierReg<0b0000, 0b101>,
|
||||
ISB = GenHintBarrierReg<0b0000, 0b110>,
|
||||
SB = GenHintBarrierReg<0b0000, 0b111>,
|
||||
};
|
||||
|
||||
// This `BarrierScope` enum is used for the dsb/dmb instructions.
|
||||
@@ -513,7 +507,7 @@ enum class SVEFMaxMinImm : uint32_t {
|
||||
_1_0,
|
||||
};
|
||||
|
||||
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
/* This `BackwardLabel` struct is used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `below` an instruction that uses it.
|
||||
* Which means that a branch would jump backwards.
|
||||
*/
|
||||
@@ -521,13 +515,11 @@ struct BackwardLabel {
|
||||
uint8_t* Location {};
|
||||
};
|
||||
|
||||
/* This `SingleUseForwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
/* This `ForwardLabel` struct is used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `above` an instruction that uses it.
|
||||
* Which means that a branch would jump forwards.
|
||||
*
|
||||
* The `ForwardLabel` struct can be bound to multiple instructions, so it needs a vector for each bind instruction type.
|
||||
*/
|
||||
struct SingleUseForwardLabel {
|
||||
struct ForwardLabel {
|
||||
enum class InstType {
|
||||
UNKNOWN,
|
||||
ADR,
|
||||
@@ -538,12 +530,16 @@ struct SingleUseForwardLabel {
|
||||
RELATIVE_LOAD,
|
||||
LONG_ADDRESS_GEN,
|
||||
};
|
||||
uint8_t* Location {};
|
||||
InstType Type = InstType::UNKNOWN;
|
||||
};
|
||||
|
||||
struct ForwardLabel {
|
||||
fextl::vector<SingleUseForwardLabel> Insts {};
|
||||
struct Reference {
|
||||
uint8_t* Location {};
|
||||
InstType Type = InstType::UNKNOWN;
|
||||
};
|
||||
|
||||
// The first element is stored separately to avoid allocations for simple cases
|
||||
Reference FirstInst;
|
||||
|
||||
fextl::vector<Reference> Insts;
|
||||
};
|
||||
|
||||
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
@@ -555,14 +551,12 @@ struct BiDirectionalLabel {
|
||||
ForwardLabel Forward;
|
||||
};
|
||||
|
||||
static inline void AddLocationToLabel(SingleUseForwardLabel* Label, SingleUseForwardLabel&& Location) {
|
||||
LOGMAN_THROW_A_FMT(Label->Type == SingleUseForwardLabel::InstType::UNKNOWN, "Trying to bind a SingleUseForwardLabel to multiple "
|
||||
"locations. Use ForwardLabel instead.");
|
||||
*Label = std::move(Location);
|
||||
}
|
||||
|
||||
static inline void AddLocationToLabel(ForwardLabel* Label, SingleUseForwardLabel&& Location) {
|
||||
Label->Insts.emplace_back(std::move(Location));
|
||||
static inline void AddLocationToLabel(ForwardLabel* Label, ForwardLabel::Reference&& Location) {
|
||||
if (Label->FirstInst.Location == nullptr) {
|
||||
Label->FirstInst = Location;
|
||||
} else {
|
||||
Label->Insts.push_back(Location);
|
||||
}
|
||||
}
|
||||
|
||||
// Some FCMA ASIMD instructions support a rotation argument.
|
||||
@@ -631,15 +625,15 @@ public:
|
||||
// Bind a backward label to an address.
|
||||
// Address that is bound is the current emitter location.
|
||||
void Bind(BackwardLabel* Label) {
|
||||
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
}
|
||||
|
||||
void Bind(const SingleUseForwardLabel* Label) {
|
||||
void Bind(const ForwardLabel::Reference* Label) {
|
||||
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
// Patch up the instructions
|
||||
switch (Label->Type) {
|
||||
case SingleUseForwardLabel::InstType::ADR: {
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
@@ -651,7 +645,7 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::ADRP: {
|
||||
case ForwardLabel::InstType::ADRP: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
@@ -665,7 +659,7 @@ public:
|
||||
break;
|
||||
}
|
||||
|
||||
case SingleUseForwardLabel::InstType::B: {
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
@@ -679,7 +673,7 @@ public:
|
||||
break;
|
||||
}
|
||||
|
||||
case SingleUseForwardLabel::InstType::TEST_BRANCH: {
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
@@ -692,8 +686,8 @@ public:
|
||||
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::BC:
|
||||
case SingleUseForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
case ForwardLabel::InstType::BC:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
@@ -705,7 +699,7 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
@@ -750,10 +744,9 @@ public:
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
template<bool WarnAboutEmpty = false>
|
||||
void Bind(ForwardLabel* Label) {
|
||||
if constexpr (WarnAboutEmpty) {
|
||||
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
|
||||
if (Label->FirstInst.Location) {
|
||||
Bind(&Label->FirstInst);
|
||||
}
|
||||
for (auto& Inst : Label->Insts) {
|
||||
Bind(&Inst);
|
||||
@@ -766,12 +759,18 @@ public:
|
||||
if (!Label->Backward.Location) {
|
||||
Bind(&Label->Backward);
|
||||
}
|
||||
Bind<false>(&Label->Forward);
|
||||
Bind(&Label->Forward);
|
||||
}
|
||||
|
||||
#include <CodeEmitter/VixlUtils.inl>
|
||||
|
||||
public:
|
||||
|
||||
// This symbol is used to allow external tooling (IDEs, clang-format, ...) to process the included files individually:
|
||||
// If defined, the files will inject member functions into this class.
|
||||
// If not, the files will wrap the member functions in a class so that tooling will process them properly.
|
||||
#define INCLUDED_BY_EMITTER
|
||||
|
||||
// TODO: Implement SME when it matters.
|
||||
#include <CodeEmitter/ALUOps.inl>
|
||||
#include <CodeEmitter/BranchOps.inl>
|
||||
@@ -781,7 +780,9 @@ public:
|
||||
#include <CodeEmitter/ASIMDOps.inl>
|
||||
#include <CodeEmitter/SVEOps.inl>
|
||||
|
||||
private:
|
||||
#undef INCLUDED_BY_EMITTER
|
||||
|
||||
protected:
|
||||
template<typename T>
|
||||
uint32_t Encode_ra(T Reg) const {
|
||||
return Reg.Idx() << 10;
|
||||
@@ -793,7 +794,6 @@ private:
|
||||
uint32_t Encode_rt2(T Reg) const {
|
||||
return Reg.Idx() << 10;
|
||||
}
|
||||
template<>
|
||||
uint32_t Encode_rt2(uint32_t Reg) const {
|
||||
return Reg << 10;
|
||||
}
|
||||
@@ -829,7 +829,6 @@ private:
|
||||
uint32_t Encode_rt(T Reg) const {
|
||||
return Reg.Idx();
|
||||
}
|
||||
template<>
|
||||
uint32_t Encode_rt(Prefetch Reg) const {
|
||||
return FEXCore::ToUnderlying(Reg);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -30,9 +30,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<Register>);
|
||||
static_assert(std::is_standard_layout_v<Register>);
|
||||
|
||||
/* 32-bit GPR register class.
|
||||
* This class will imply a 32-bit register size being used.
|
||||
@@ -58,9 +58,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(WRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<WRegister>);
|
||||
static_assert(std::is_standard_layout_v<WRegister>);
|
||||
|
||||
/* 64-bit GPR register class.
|
||||
* This class will imply a 64-bit register size being used.
|
||||
@@ -86,9 +86,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(XRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<XRegister>);
|
||||
static_assert(std::is_standard_layout_v<XRegister>);
|
||||
|
||||
inline constexpr WRegister Register::W() const {
|
||||
return WRegister {Index};
|
||||
@@ -283,9 +283,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(VRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<VRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<VRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(VRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<VRegister>);
|
||||
static_assert(std::is_standard_layout_v<VRegister>);
|
||||
|
||||
/* 8-bit ASIMD register class
|
||||
* This class implies 8-bit scalar register.
|
||||
@@ -315,9 +315,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(BRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<BRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<BRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(BRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<BRegister>);
|
||||
static_assert(std::is_standard_layout_v<BRegister>);
|
||||
|
||||
/* 16-bit ASIMD register class
|
||||
* This class implies 16-bit scalar register.
|
||||
@@ -347,9 +347,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(HRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<HRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<HRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(HRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<HRegister>);
|
||||
static_assert(std::is_standard_layout_v<HRegister>);
|
||||
|
||||
/* 32-bit ASIMD register class
|
||||
* This class implies 32-bit scalar register.
|
||||
@@ -379,9 +379,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(SRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<SRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<SRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(SRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<SRegister>);
|
||||
static_assert(std::is_standard_layout_v<SRegister>);
|
||||
|
||||
/* 64-bit ASIMD register class
|
||||
* This class doesn't imply Vector or Scalar.
|
||||
@@ -412,9 +412,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(DRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<DRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<DRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(DRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<DRegister>);
|
||||
static_assert(std::is_standard_layout_v<DRegister>);
|
||||
|
||||
/* 128-bit ASIMD register class
|
||||
* This class doesn't imply Vector or Scalar.
|
||||
@@ -445,9 +445,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(QRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<QRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<QRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(QRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<QRegister>);
|
||||
static_assert(std::is_standard_layout_v<QRegister>);
|
||||
|
||||
/* Unsized SVE register class.
|
||||
* This class explicitly implies the instruction will operate using SVE.
|
||||
@@ -474,9 +474,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(ZRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<ZRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<ZRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(ZRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<ZRegister>);
|
||||
static_assert(std::is_standard_layout_v<ZRegister>);
|
||||
|
||||
// VRegister
|
||||
inline constexpr BRegister VRegister::B() const {
|
||||
@@ -919,9 +919,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegister>);
|
||||
static_assert(std::is_standard_layout_v<PRegister>);
|
||||
|
||||
// Unsized predicate register for SVE with zeroing semantics.
|
||||
class PRegisterZero {
|
||||
@@ -947,9 +947,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegisterZero>);
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>);
|
||||
|
||||
// Unsized predicate register for SVE with merging semantics.
|
||||
class PRegisterMerge {
|
||||
@@ -975,9 +975,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegisterMerge) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegisterMerge>);
|
||||
static_assert(std::is_standard_layout_v<PRegisterMerge>);
|
||||
|
||||
// PRegister
|
||||
inline constexpr PRegisterZero PRegister::Zeroing() const {
|
||||
|
||||
+501
-653
File diff suppressed because it is too large.
Load diff
@@ -16,17 +16,25 @@
|
||||
* Exceptions to this rule will have asserts in the emitter implementation when misused.
|
||||
*
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
public:
|
||||
// Advanced SIMD scalar copy
|
||||
// Advanced SIMD scalar copy
|
||||
void dup(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Index) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'01 << 10;
|
||||
|
||||
const uint32_t SizeImm = FEXCore::ToUnderlying(size);
|
||||
const uint32_t IndexShift = SizeImm + 1;
|
||||
const uint32_t ElementSize = 1U << SizeImm;
|
||||
const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
[[maybe_unused]] const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
|
||||
LOGMAN_THROW_A_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
|
||||
|
||||
const uint32_t imm5 = (Index << IndexShift) | ElementSize;
|
||||
|
||||
@@ -37,7 +45,7 @@ public:
|
||||
dup(size, rd, rn, Index);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same FP16
|
||||
// Advanced SIMD scalar three same FP16
|
||||
void fmulx(HRegister rd, HRegister rn, HRegister rm) {
|
||||
ASIMDScalarThreeSameFP16(0, 0, 0b011, rm, rn, rd);
|
||||
}
|
||||
@@ -66,7 +74,7 @@ public:
|
||||
ASIMDScalarThreeSameFP16(1, 1, 0b101, rm, rn, rd);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
void fcvtns(HRegister rd, HRegister rn) {
|
||||
ASIMDScalarTwoRegMiscFP16(0, 0, 0b11010, rn, rd);
|
||||
}
|
||||
@@ -128,9 +136,9 @@ public:
|
||||
ASIMDScalarTwoRegMiscFP16(1, 1, 0b11101, rn, rd);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
void suqadd(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b00011, rd, rn);
|
||||
}
|
||||
@@ -140,67 +148,55 @@ public:
|
||||
|
||||
///< Comparison against 0.0
|
||||
void cmgt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01000, rd, rn);
|
||||
}
|
||||
///< Comparison against 0.0
|
||||
void cmeq(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01001, rd, rn);
|
||||
}
|
||||
|
||||
///< Comparison against 0.0
|
||||
void cmlt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01010, rd, rn);
|
||||
}
|
||||
void abs(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01011, rd, rn);
|
||||
}
|
||||
///< size is destination size.
|
||||
void sqxtn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b10100, rd, rn);
|
||||
}
|
||||
|
||||
void fcvtns(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
void fcvtms(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
void fcvtas(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
void scvtf(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -249,70 +245,58 @@ public:
|
||||
}
|
||||
///< Comparison against 0.0
|
||||
void cmge(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b01000, rd, rn);
|
||||
}
|
||||
///< Comparison against 0.0
|
||||
void cmle(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b01001, rd, rn);
|
||||
}
|
||||
void neg(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b01011, rd, rn);
|
||||
}
|
||||
///< size is destination.
|
||||
void sqxtun(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b10010, rd, rn);
|
||||
}
|
||||
///< size is destination.
|
||||
void uqxtn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b10100, rd, rn);
|
||||
}
|
||||
///< size is destination.
|
||||
void fcvtxn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMDScalar2RegMisc(0, 1, ScalarRegSize::i16Bit, 0b10110, rd, rn);
|
||||
}
|
||||
void fcvtnu(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
void fcvtmu(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
void fcvtau(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
void ucvtf(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -366,73 +350,55 @@ public:
|
||||
}
|
||||
|
||||
void fmaxnmp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(1, 1, ConvertedSize, 0b01100, rd, rn);
|
||||
}
|
||||
void faddp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(1, 1, ConvertedSize, 0b01101, rd, rn);
|
||||
}
|
||||
void fmaxp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(1, 1, ConvertedSize, 0b01111, rd, rn);
|
||||
}
|
||||
void fminnmp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMDScalar2RegMisc(1, 1, size, 0b01100, rd, rn);
|
||||
}
|
||||
void fminp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMDScalar2RegMisc(1, 1, size, 0b01111, rd, rn);
|
||||
}
|
||||
// Advanced SIMD scalar three different
|
||||
// Advanced SIMD scalar three different
|
||||
///< size is destination.
|
||||
void sqdmlal(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i32Bit :
|
||||
ScalarRegSize::i16Bit;
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i32Bit : ScalarRegSize::i16Bit;
|
||||
ASIMD3RegDifferent(0, ConvertedSize, 0b1001, rd, rn, rm);
|
||||
}
|
||||
///< size is destination.
|
||||
void sqdmlsl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i32Bit :
|
||||
ScalarRegSize::i16Bit;
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i32Bit : ScalarRegSize::i16Bit;
|
||||
ASIMD3RegDifferent(0, ConvertedSize, 0b1011, rd, rn, rm);
|
||||
}
|
||||
|
||||
///< size is destination.
|
||||
void sqdmull(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i32Bit :
|
||||
ScalarRegSize::i16Bit;
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i32Bit : ScalarRegSize::i16Bit;
|
||||
ASIMD3RegDifferent(0, ConvertedSize, 0b1101, rd, rn, rm);
|
||||
}
|
||||
// Advanced SIMD scalar three same
|
||||
// Advanced SIMD scalar three same
|
||||
void sqadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(0, size, 0b00001, rd, rn, rm);
|
||||
}
|
||||
@@ -440,71 +406,62 @@ public:
|
||||
ASIMD3RegSame(0, size, 0b00101, rd, rn, rm);
|
||||
}
|
||||
void cmgt(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b00110, rd, rn, rm);
|
||||
}
|
||||
void cmge(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b00111, rd, rn, rm);
|
||||
}
|
||||
void sshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b01000, rd, rn, rm);
|
||||
}
|
||||
void sqshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(0, size, 0b01001, rd, rn, rm);
|
||||
}
|
||||
void srshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b01010, rd, rn, rm);
|
||||
}
|
||||
void sqrshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(0, size, 0b01011, rd, rn, rm);
|
||||
}
|
||||
void add(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b10000, rd, rn, rm);
|
||||
}
|
||||
void cmtst(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b10001, rd, rn, rm);
|
||||
}
|
||||
void sqdmulh(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
ASIMD3RegSame(0, size, 0b10110, rd, rn, rm);
|
||||
}
|
||||
void fmulx(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(0, ConvertedSize, 0b11011, rd, rn, rm);
|
||||
}
|
||||
void fcmeq(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(0, ConvertedSize, 0b11100, rd, rn, rm);
|
||||
}
|
||||
void frecps(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(0, ConvertedSize, 0b11111, rd, rn, rm);
|
||||
}
|
||||
void frsqrts(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(0, size, 0b11111, rd, rn, rm);
|
||||
}
|
||||
void uqadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
@@ -514,75 +471,69 @@ public:
|
||||
ASIMD3RegSame(1, size, 0b00101, rd, rn, rm);
|
||||
}
|
||||
void cmhi(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b00110, rd, rn, rm);
|
||||
}
|
||||
void cmhs(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b00111, rd, rn, rm);
|
||||
}
|
||||
void ushl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b01000, rd, rn, rm);
|
||||
}
|
||||
void uqshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(1, size, 0b01001, rd, rn, rm);
|
||||
}
|
||||
void urshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b01010, rd, rn, rm);
|
||||
}
|
||||
void uqrshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(1, size, 0b01011, rd, rn, rm);
|
||||
}
|
||||
void sub(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b10000, rd, rn, rm);
|
||||
}
|
||||
void cmeq(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b10001, rd, rn, rm);
|
||||
}
|
||||
void sqrdmulh(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
ASIMD3RegSame(1, size, 0b10110, rd, rn, rm);
|
||||
}
|
||||
void fcmge(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(1, ConvertedSize, 0b11100, rd, rn, rm);
|
||||
}
|
||||
void facge(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(1, ConvertedSize, 0b11101, rd, rn, rm);
|
||||
}
|
||||
void fabd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(1, size, 0b11010, rd, rn, rm);
|
||||
}
|
||||
void fcmgt(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(1, size, 0b11100, rd, rn, rm);
|
||||
}
|
||||
void facgt(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(1, size, 0b11101, rd, rn, rm);
|
||||
}
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
void sshr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -592,8 +543,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00000, rd, rn);
|
||||
}
|
||||
void ssra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -603,8 +554,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00010, rd, rn);
|
||||
}
|
||||
void srshr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -614,8 +565,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00100, rd, rn);
|
||||
}
|
||||
void srsra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -625,8 +576,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00110, rd, rn);
|
||||
}
|
||||
void shl(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
// Shift encoded a bit weirdly.
|
||||
// shift = immh:immb - elementsize but immh is /also/ used for element size.
|
||||
const uint32_t immh = 1 << FEXCore::ToUnderlying(size) | (Shift >> 3);
|
||||
@@ -644,7 +595,7 @@ public:
|
||||
///< size is destination
|
||||
void sqshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -655,7 +606,7 @@ public:
|
||||
}
|
||||
void sqrshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -666,8 +617,8 @@ public:
|
||||
}
|
||||
// TODO: SCVTF, FCVTZS
|
||||
void ushr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -677,8 +628,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00000, rd, rn);
|
||||
}
|
||||
void usra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -688,8 +639,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00010, rd, rn);
|
||||
}
|
||||
void urshr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -699,8 +650,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00100, rd, rn);
|
||||
}
|
||||
void ursra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -710,8 +661,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00110, rd, rn);
|
||||
}
|
||||
void sri(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -721,8 +672,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b01000, rd, rn);
|
||||
}
|
||||
void sli(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
// Shift encoded a bit weirdly.
|
||||
// shift = immh:immb - elementsize but immh is /also/ used for element size.
|
||||
const uint32_t immh = 1 << FEXCore::ToUnderlying(size) | (Shift >> 3);
|
||||
@@ -748,7 +699,7 @@ public:
|
||||
///< size is destination.
|
||||
void sqshrun(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -760,7 +711,7 @@ public:
|
||||
///< size is destination.
|
||||
void sqrshrun(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -772,7 +723,7 @@ public:
|
||||
///< size is destination.
|
||||
void uqshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -784,7 +735,7 @@ public:
|
||||
///< size is destination.
|
||||
void uqrshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -794,10 +745,10 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b10011, rd, rn);
|
||||
}
|
||||
// TODO: UCVTF, FCVTZU
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
void fmov(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000000, rd, rn);
|
||||
}
|
||||
@@ -991,14 +942,14 @@ public:
|
||||
Float1Source(0, 0, 0b11, 0b001111, rd.V(), rn.V());
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
// Floating-point compare
|
||||
void fcmp(ScalarRegSize Size, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(Size != ScalarRegSize::i8Bit, "8-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(Size != ScalarRegSize::i8Bit, "8-bit destination not supported");
|
||||
|
||||
const auto ConvertedSize =
|
||||
Size == ARMEmitter::ScalarRegSize::i64Bit ? 0b01 :
|
||||
Size == ARMEmitter::ScalarRegSize::i32Bit ? 0b00 :
|
||||
Size == ARMEmitter::ScalarRegSize::i16Bit ? 0b11 : 0;
|
||||
const auto ConvertedSize = Size == ARMEmitter::ScalarRegSize::i64Bit ? 0b01 :
|
||||
Size == ARMEmitter::ScalarRegSize::i32Bit ? 0b00 :
|
||||
Size == ARMEmitter::ScalarRegSize::i16Bit ? 0b11 :
|
||||
0;
|
||||
|
||||
FloatCompare(0, 0, ConvertedSize, 0b00, 0b00000, rn, rm);
|
||||
}
|
||||
@@ -1051,7 +1002,7 @@ public:
|
||||
FloatCompare(0, 0, 0b11, 0b00, 0b11000, rn.V(), VReg::v0);
|
||||
}
|
||||
|
||||
// Floating-point immediate
|
||||
// Floating-point immediate
|
||||
void fmov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, float Value) {
|
||||
uint32_t M = 0;
|
||||
uint32_t S = 0;
|
||||
@@ -1061,16 +1012,13 @@ public:
|
||||
if (size == ARMEmitter::ScalarRegSize::i16Bit) {
|
||||
LOGMAN_MSG_A_FMT("Unsupported");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
else if (size == ARMEmitter::ScalarRegSize::i32Bit) {
|
||||
} else if (size == ARMEmitter::ScalarRegSize::i32Bit) {
|
||||
ptype = 0b00;
|
||||
imm8 = FP32ToImm8(Value);
|
||||
}
|
||||
else if (size == ARMEmitter::ScalarRegSize::i64Bit) {
|
||||
} else if (size == ARMEmitter::ScalarRegSize::i64Bit) {
|
||||
ptype = 0b01;
|
||||
imm8 = FP64ToImm8(Value);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
@@ -1090,7 +1038,7 @@ public:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Floating-point conditional compare
|
||||
// Floating-point conditional compare
|
||||
void fccmp(SRegister rn, SRegister rm, StatusFlags flags, Condition Cond) {
|
||||
FloatConditionalCompare(0, 0, 0b00, 0b0, rn.V(), rm.V(), flags, Cond);
|
||||
}
|
||||
@@ -1110,7 +1058,7 @@ public:
|
||||
FloatConditionalCompare(0, 0, 0b11, 0b1, rn.V(), rm.V(), flags, Cond);
|
||||
}
|
||||
|
||||
// Floating-point data-processing (2 source)
|
||||
// Floating-point data-processing (2 source)
|
||||
void fmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0000, rd, rn, rm);
|
||||
}
|
||||
@@ -1225,11 +1173,10 @@ public:
|
||||
|
||||
// Floating-point conditional select
|
||||
void fcsel(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit,
|
||||
"Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
const uint32_t ConvertedSize = size == ScalarRegSize::i64Bit ? 0b01 : size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
FloatConditionalSelect(0, 0, ConvertedSize, rd, rn, rm, Cond);
|
||||
}
|
||||
@@ -1244,7 +1191,7 @@ public:
|
||||
FloatConditionalSelect(0, 0, 0b11, rd.V(), rn.V(), rm.V(), Cond);
|
||||
}
|
||||
|
||||
// Floating-point data-processing (3 source)
|
||||
// Floating-point data-processing (3 source)
|
||||
void fmadd(SRegister rd, SRegister rn, SRegister rm, SRegister ra) {
|
||||
Float3Source(0, 0, 0b00, 0, 0, rd.V(), rn.V(), rm.V(), ra.V());
|
||||
}
|
||||
@@ -1285,7 +1232,7 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
// Advanced SIMD scalar copy
|
||||
// Advanced SIMD scalar copy
|
||||
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -1297,7 +1244,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same FP16
|
||||
// Advanced SIMD scalar three same FP16
|
||||
void ASIMDScalarThreeSameFP16(uint32_t U, uint32_t a, uint32_t opcode, HRegister rm, HRegister rn, HRegister rd) {
|
||||
uint32_t Instr = 0b0101'1110'0100'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1309,7 +1256,7 @@ private:
|
||||
Instr |= rd.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
void ASIMDScalarTwoRegMiscFP16(uint32_t U, uint32_t a, uint32_t opcode, HRegister rn, HRegister rd) {
|
||||
uint32_t Instr = 0b0101'1110'0111'1000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1321,9 +1268,9 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
void ASIMDScalar2RegMisc(uint32_t b20, uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
uint32_t Instr = 0b0101'1110'0010'0000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1336,9 +1283,9 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar pairwise
|
||||
// XXX:
|
||||
// Advanced SIMD scalar three different
|
||||
// Advanced SIMD scalar pairwise
|
||||
// XXX:
|
||||
// Advanced SIMD scalar three different
|
||||
void ASIMD3RegDifferent(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0101'1110'0010'0000'0000'0000'0000'0000;
|
||||
|
||||
@@ -1350,7 +1297,7 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar three same
|
||||
// Advanced SIMD scalar three same
|
||||
void ASIMD3RegSame(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0101'1110'0010'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1362,7 +1309,7 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
void ASIMDScalarShiftByImm(uint32_t U, uint32_t immh, uint32_t immb, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
uint32_t Instr = 0b0101'1111'0000'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1374,9 +1321,9 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
// Floating-point data-processing (1 source)
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
// Floating-point data-processing (1 source)
|
||||
void Float1Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0100'0000'0000'0000;
|
||||
|
||||
@@ -1390,16 +1337,15 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
void Float1Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit,
|
||||
"Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
const uint32_t ConvertedSize = size == ScalarRegSize::i64Bit ? 0b01 : size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float1Source(M, S, ConvertedSize, opcode, rd, rn);
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
// Floating-point compare
|
||||
void FloatCompare(uint32_t M, uint32_t S, uint32_t ftype, uint32_t op, uint32_t opcode2, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0010'0000'0000'0000;
|
||||
|
||||
@@ -1413,9 +1359,9 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point immediate
|
||||
// XXX:
|
||||
// Floating-point conditional compare
|
||||
// Floating-point immediate
|
||||
// XXX:
|
||||
// Floating-point conditional compare
|
||||
void FloatConditionalCompare(uint32_t M, uint32_t S, uint32_t ptype, uint32_t op, VRegister rn, VRegister rm, StatusFlags flags, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1430,7 +1376,7 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point data-processing (2 source)
|
||||
// Floating-point data-processing (2 source)
|
||||
|
||||
void Float2Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1000'0000'0000;
|
||||
@@ -1447,16 +1393,15 @@ private:
|
||||
}
|
||||
|
||||
void Float2Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit,
|
||||
"Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
const uint32_t ConvertedSize = size == ScalarRegSize::i64Bit ? 0b01 : size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float2Source(M, S, ConvertedSize, opcode, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
// Floating-point conditional select
|
||||
void FloatConditionalSelect(uint32_t M, uint32_t S, uint32_t ptype, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1100'0000'0000;
|
||||
|
||||
@@ -1470,7 +1415,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Floating-point data-processing (3 source)
|
||||
// Floating-point data-processing (3 source)
|
||||
void Float3Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t o1, uint32_t o0, VRegister rd, VRegister rn, VRegister rm, VRegister ra) {
|
||||
uint32_t Instr = 0b0001'1111'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
@@ -1485,3 +1430,8 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
@@ -4,173 +4,185 @@
|
||||
* This is mostly a mashup of various instruction types.
|
||||
* Nothing follows an explicit pattern since they are mostly different.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
public:
|
||||
// System with result
|
||||
// TODO: SYSL
|
||||
// System Instruction
|
||||
// TODO: AT
|
||||
// TODO: CFP
|
||||
// TODO: CPP
|
||||
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
|
||||
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
|
||||
}
|
||||
// TODO: DVP
|
||||
// TODO: IC
|
||||
// TODO: TLBI
|
||||
// System with result
|
||||
// TODO: SYSL
|
||||
// System Instruction
|
||||
// TODO: AT
|
||||
// TODO: CFP
|
||||
// TODO: CPP
|
||||
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
|
||||
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
|
||||
}
|
||||
// TODO: DVP
|
||||
// TODO: IC
|
||||
// TODO: TLBI
|
||||
|
||||
// Exception generation
|
||||
void svc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
|
||||
}
|
||||
void hvc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
|
||||
}
|
||||
void smc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
|
||||
}
|
||||
void brk(uint32_t Imm) {
|
||||
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
|
||||
}
|
||||
void hlt(uint32_t Imm) {
|
||||
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
|
||||
}
|
||||
void tcancel(uint32_t Imm) {
|
||||
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
|
||||
}
|
||||
void dcps1(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
|
||||
}
|
||||
void dcps2(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
|
||||
}
|
||||
void dcps3(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
|
||||
}
|
||||
// System instructions with register argument
|
||||
void wfet(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b000, rt);
|
||||
}
|
||||
void wfit(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b001, rt);
|
||||
}
|
||||
// Exception generation
|
||||
void svc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
|
||||
}
|
||||
void hvc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
|
||||
}
|
||||
void smc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
|
||||
}
|
||||
void brk(uint32_t Imm) {
|
||||
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
|
||||
}
|
||||
void hlt(uint32_t Imm) {
|
||||
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
|
||||
}
|
||||
void tcancel(uint32_t Imm) {
|
||||
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
|
||||
}
|
||||
void dcps1(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
|
||||
}
|
||||
void dcps2(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
|
||||
}
|
||||
void dcps3(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
|
||||
}
|
||||
// System instructions with register argument
|
||||
void wfet(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b000, rt);
|
||||
}
|
||||
void wfit(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b001, rt);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void nop() {
|
||||
Hint(ARMEmitter::HintRegister::NOP);
|
||||
}
|
||||
void yield() {
|
||||
Hint(ARMEmitter::HintRegister::YIELD);
|
||||
}
|
||||
void wfe() {
|
||||
Hint(ARMEmitter::HintRegister::WFE);
|
||||
}
|
||||
void wfi() {
|
||||
Hint(ARMEmitter::HintRegister::WFI);
|
||||
}
|
||||
void sev() {
|
||||
Hint(ARMEmitter::HintRegister::SEV);
|
||||
}
|
||||
void sevl() {
|
||||
Hint(ARMEmitter::HintRegister::SEVL);
|
||||
}
|
||||
void dgh() {
|
||||
Hint(ARMEmitter::HintRegister::DGH);
|
||||
}
|
||||
void csdb() {
|
||||
Hint(ARMEmitter::HintRegister::CSDB);
|
||||
}
|
||||
// Hints
|
||||
void nop() {
|
||||
Hint(ARMEmitter::HintRegister::NOP);
|
||||
}
|
||||
void yield() {
|
||||
Hint(ARMEmitter::HintRegister::YIELD);
|
||||
}
|
||||
void wfe() {
|
||||
Hint(ARMEmitter::HintRegister::WFE);
|
||||
}
|
||||
void wfi() {
|
||||
Hint(ARMEmitter::HintRegister::WFI);
|
||||
}
|
||||
void sev() {
|
||||
Hint(ARMEmitter::HintRegister::SEV);
|
||||
}
|
||||
void sevl() {
|
||||
Hint(ARMEmitter::HintRegister::SEVL);
|
||||
}
|
||||
void dgh() {
|
||||
Hint(ARMEmitter::HintRegister::DGH);
|
||||
}
|
||||
void csdb() {
|
||||
Hint(ARMEmitter::HintRegister::CSDB);
|
||||
}
|
||||
|
||||
// Barriers
|
||||
void clrex(uint32_t imm = 15) {
|
||||
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
|
||||
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
}
|
||||
void dsb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void dmb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void isb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
|
||||
}
|
||||
void sb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::SB, 0);
|
||||
}
|
||||
void tcommit() {
|
||||
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
}
|
||||
// Barriers
|
||||
void clrex(uint32_t imm = 15) {
|
||||
LOGMAN_THROW_A_FMT(imm < 16, "Immediate out of range");
|
||||
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
}
|
||||
void dsb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void dmb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void isb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
|
||||
}
|
||||
void sb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::SB, 0);
|
||||
}
|
||||
void tcommit() {
|
||||
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
|
||||
SystemRegisterMove(Op, rt, reg);
|
||||
}
|
||||
// System register move
|
||||
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
|
||||
SystemRegisterMove(Op, rt, reg);
|
||||
}
|
||||
|
||||
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
|
||||
SystemRegisterMove(Op, rd, reg);
|
||||
}
|
||||
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
|
||||
SystemRegisterMove(Op, rd, reg);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
// Exception Generation
|
||||
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
|
||||
LOGMAN_THROW_AA_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
|
||||
// Exception Generation
|
||||
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
|
||||
|
||||
uint32_t Instr = 0b1101'0100 << 24;
|
||||
uint32_t Instr = 0b1101'0100 << 24;
|
||||
|
||||
Instr |= opc << 21;
|
||||
Instr |= Imm << 5;
|
||||
Instr |= op2 << 2;
|
||||
Instr |= LL;
|
||||
Instr |= opc << 21;
|
||||
Instr |= Imm << 5;
|
||||
Instr |= op2 << 2;
|
||||
Instr |= LL;
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System instructions with register argument
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
|
||||
// System instructions with register argument
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
|
||||
|
||||
Instr |= CRm << 8;
|
||||
Instr |= op2 << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
Instr |= CRm << 8;
|
||||
Instr |= op2 << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void Hint(ARMEmitter::HintRegister Reg) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Barriers
|
||||
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
|
||||
Instr |= CRm << 8;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Hints
|
||||
void Hint(ARMEmitter::HintRegister Reg) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Barriers
|
||||
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
|
||||
Instr |= CRm << 8;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System Instruction
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = Op;
|
||||
// System Instruction
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= L << 21;
|
||||
Instr |= SubOp;
|
||||
Instr |= Encode_rt(rt);
|
||||
Instr |= L << 21;
|
||||
Instr |= SubOp;
|
||||
Instr |= Encode_rt(rt);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
|
||||
uint32_t Instr = Op;
|
||||
// System register move
|
||||
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= FEXCore::ToUnderlying(reg);
|
||||
Instr |= Encode_rt(rt);
|
||||
Instr |= FEXCore::ToUnderlying(reg);
|
||||
Instr |= Encode_rt(rt);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
@@ -34,11 +34,7 @@
|
||||
// by the corresponding fields in the logical instruction.
|
||||
// If it can not be encoded, the function returns false, and the values pointed
|
||||
// to by n, imm_s and imm_r are undefined.
|
||||
static bool IsImmLogical(uint64_t value,
|
||||
unsigned width,
|
||||
unsigned* n = nullptr,
|
||||
unsigned* imm_s = nullptr,
|
||||
unsigned* imm_r = nullptr) {
|
||||
static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr, unsigned* imm_s = nullptr, unsigned* imm_r = nullptr) {
|
||||
[[maybe_unused]] constexpr auto kBRegSize = 8;
|
||||
[[maybe_unused]] constexpr auto kHRegSize = 16;
|
||||
[[maybe_unused]] constexpr auto kSRegSize = 32;
|
||||
@@ -47,8 +43,7 @@ static bool IsImmLogical(uint64_t value,
|
||||
constexpr auto kWRegSize = 32;
|
||||
constexpr auto kXRegSize = 64;
|
||||
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) ||
|
||||
(width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) || (width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
|
||||
bool negate = false;
|
||||
|
||||
@@ -182,12 +177,7 @@ static bool IsImmLogical(uint64_t value,
|
||||
// (1 + 2^d + 2^(2d) + ...), i.e. 0x0001000100010001 or similar. These can
|
||||
// be derived using a table lookup on CLZ(d).
|
||||
static const uint64_t multipliers[] = {
|
||||
0x0000000000000001UL,
|
||||
0x0000000100000001UL,
|
||||
0x0001000100010001UL,
|
||||
0x0101010101010101UL,
|
||||
0x1111111111111111UL,
|
||||
0x5555555555555555UL,
|
||||
0x0000000000000001UL, 0x0000000100000001UL, 0x0001000100010001UL, 0x0101010101010101UL, 0x1111111111111111UL, 0x5555555555555555UL,
|
||||
};
|
||||
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
|
||||
uint64_t candidate = (b - a) * multiplier;
|
||||
@@ -244,7 +234,9 @@ static bool IsImmLogical(uint64_t value,
|
||||
}
|
||||
|
||||
static inline bool IsIntN(unsigned n, int64_t x) {
|
||||
if (n == 64) return true;
|
||||
if (n == 64) {
|
||||
return true;
|
||||
}
|
||||
int64_t limit = INT64_C(1) << (n - 1);
|
||||
return (-limit <= x) && (x < limit);
|
||||
}
|
||||
@@ -271,11 +263,15 @@ V(57) V(58) V(59) V(60) V(61) V(62) V(63)
|
||||
|
||||
// clang-format on
|
||||
|
||||
#define DECLARE_IS_INT_N(N) \
|
||||
static inline bool IsInt##N(int64_t x) { return IsIntN(N, x); }
|
||||
#define DECLARE_IS_INT_N(N) \
|
||||
static inline bool IsInt##N(int64_t x) { \
|
||||
return IsIntN(N, x); \
|
||||
}
|
||||
|
||||
#define DECLARE_IS_UINT_N(N) \
|
||||
static inline bool IsUint##N(int64_t x) { return IsUintN(N, x); }
|
||||
#define DECLARE_IS_UINT_N(N) \
|
||||
static inline bool IsUint##N(int64_t x) { \
|
||||
return IsUintN(N, x); \
|
||||
}
|
||||
|
||||
INT_1_TO_63_LIST(DECLARE_IS_INT_N)
|
||||
INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
@@ -285,14 +281,14 @@ INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
|
||||
private:
|
||||
|
||||
template <typename V>
|
||||
template<typename V>
|
||||
static inline bool IsPowerOf2(V value) {
|
||||
return (value != 0) && ((value & (value - 1)) == 0);
|
||||
}
|
||||
|
||||
// Some compilers dislike negating unsigned integers,
|
||||
// so we provide an equivalent.
|
||||
template <typename T>
|
||||
template<typename T>
|
||||
static inline T UnsignedNegate(T value) {
|
||||
static_assert(std::is_unsigned<T>::value);
|
||||
return ~value + 1;
|
||||
@@ -302,7 +298,7 @@ static inline uint64_t LowestSetBit(uint64_t value) {
|
||||
return value & UnsignedNegate(value);
|
||||
}
|
||||
|
||||
template <typename V>
|
||||
template<typename V>
|
||||
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
|
||||
#if COMPILER_HAS_BUILTIN_CLZ
|
||||
if (width == 32) {
|
||||
|
||||
+1
-136
@@ -9,25 +9,6 @@
|
||||
"@PREFIX_LIB@/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
"Library": "libGLESv2-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libGLESv2.so",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libX11.so",
|
||||
"@PREFIX_LIB@/libX11.so.6",
|
||||
"@PREFIX_LIB@/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Overlay": [
|
||||
@@ -36,89 +17,6 @@
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb.so",
|
||||
"@PREFIX_LIB@/libxcb.so.1",
|
||||
"@PREFIX_LIB@/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-present.so",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxshmfence.so",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
@@ -141,38 +39,6 @@
|
||||
"@PREFIX_LIB@/libfex_thunk_test.so"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXrender.so",
|
||||
"@PREFIX_LIB@/libXrender.so.1",
|
||||
"@PREFIX_LIB@/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXext.so",
|
||||
"@PREFIX_LIB@/libXext.so.6",
|
||||
"@PREFIX_LIB@/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXfixes.so",
|
||||
"@PREFIX_LIB@/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libOpenCL.so",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
@@ -180,7 +46,6 @@
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
}
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 29f979ee5a...cacef3039d.
Vendored
+1
-1
Submodule External/drm-headers updated: 8efb6dc03f...0675d2f291.
Vendored
+1
-1
Submodule External/fmt updated: 0c9fce2ffe...123913715a.
Vendored
+1
-1
@@ -1,3 +1,3 @@
|
||||
set(NAME tiny-json)
|
||||
set(SRCS tiny-json.c)
|
||||
add_library(${NAME} ${SRCS})
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
+1
Submodule External/tracy added at 5d542dc09f.
Vendored
+1
-1
Submodule External/vixl updated: a90f5d5020...84bc10c107.
Vendored
-1
Submodule External/xbyak deleted from c68cc53d18.
@@ -188,27 +188,33 @@ def print_man_environment_tail():
|
||||
|
||||
# Additional environment variables that live outside of the normal loop
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG_LOCATION",
|
||||
"APP_CONFIG_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for configuration files",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG",
|
||||
"APP_CONFIG",
|
||||
[
|
||||
"Allows the user to override where FEX looks for only the application config file",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
|
||||
"This will override this file location",
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_DATA_LOCATION",
|
||||
"APP_DATA_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for data files",
|
||||
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
|
||||
@@ -218,9 +224,13 @@ def print_man_environment_tail():
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
"PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored. These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For Arm64ec/Wow64 WINE builds:",
|
||||
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
],
|
||||
"''", True)
|
||||
@@ -413,7 +423,7 @@ def print_parse_argloader_options(options):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
output_argloader.write("\t\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
@@ -451,14 +461,19 @@ def print_parse_jsonloader_options(options):
|
||||
output_argloader.write("#ifdef JSONLOADER\n")
|
||||
output_argloader.write("#undef JSONLOADER\n")
|
||||
output_argloader.write("if (false) {}\n")
|
||||
op_key = None
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
elif (value_type == "strarray"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
assert op_key is not None, "No options found in JSONLOADER"
|
||||
output_argloader.write("else {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
|
||||
@@ -44,7 +44,7 @@ class OpDefinition:
|
||||
HasDest: bool
|
||||
DestType: str
|
||||
DestSize: str
|
||||
NumElements: str
|
||||
ElementSize: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
@@ -67,7 +67,7 @@ class OpDefinition:
|
||||
self.HasDest = False
|
||||
self.DestType = None
|
||||
self.DestSize = None
|
||||
self.NumElements = None
|
||||
self.ElementSize = None
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
@@ -232,8 +232,8 @@ def parse_ops(ops):
|
||||
if "DestSize" in op_val:
|
||||
OpDef.DestSize = op_val["DestSize"]
|
||||
|
||||
if "NumElements" in op_val:
|
||||
OpDef.NumElements = op_val["NumElements"]
|
||||
if "ElementSize" in op_val:
|
||||
OpDef.ElementSize = op_val["ElementSize"]
|
||||
|
||||
if len(op_class):
|
||||
OpDef.OpClass = op_class
|
||||
@@ -374,7 +374,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("};\n")
|
||||
|
||||
# Add a static assert that the IR ops must be pod
|
||||
output_file.write("static_assert(std::is_trivial_v<IROp_{}>);\n".format(op.Name))
|
||||
output_file.write("static_assert(std::is_trivially_copyable_v<IROp_{}>);\n".format(op.Name))
|
||||
output_file.write("static_assert(std::is_standard_layout_v<IROp_{}>);\n\n".format(op.Name))
|
||||
|
||||
output_file.write("#undef IROP_STRUCTS\n")
|
||||
@@ -743,10 +743,10 @@ def print_ir_allocator_helpers():
|
||||
if op.DestSize != None:
|
||||
output_file.write("\t\t_Op.first->Header.Size = {};\n".format(op.DestSize))
|
||||
|
||||
if op.NumElements == None:
|
||||
if op.ElementSize == None:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size;\n")
|
||||
else:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = {};\n".format(op.ElementSize))
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
|
||||
@@ -24,9 +24,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/extF80_mul.c
|
||||
Common/SoftFloat-3e/extF80_rem.c
|
||||
Common/SoftFloat-3e/extF80_sqrt.c
|
||||
Common/SoftFloat-3e/s_add128.c
|
||||
Common/SoftFloat-3e/s_sub128.c
|
||||
Common/SoftFloat-3e/s_le128.c
|
||||
Common/SoftFloat-3e/extF80_to_i32.c
|
||||
Common/SoftFloat-3e/extF80_to_i64.c
|
||||
Common/SoftFloat-3e/extF80_to_ui64.c
|
||||
@@ -39,7 +36,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_roundToUI64.c
|
||||
Common/SoftFloat-3e/s_f128UIToCommonNaN.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF128UI.c
|
||||
Common/SoftFloat-3e/s_shortShiftRight128.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF128Sig.c
|
||||
Common/SoftFloat-3e/s_roundToI32.c
|
||||
Common/SoftFloat-3e/s_roundToI64.c
|
||||
@@ -48,22 +44,14 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_extF80UIToCommonNaN.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF32UI.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF64UI.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_roundPackToF64.c
|
||||
Common/SoftFloat-3e/s_propagateNaNExtF80UI.c
|
||||
Common/SoftFloat-3e/s_roundPackToExtF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalExtF80Sig.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam64.c
|
||||
Common/SoftFloat-3e/s_subMagsExtF80.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam32.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam128.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam128Extra.c
|
||||
Common/SoftFloat-3e/s_normRoundPackToExtF80.c
|
||||
Common/SoftFloat-3e/s_shortShiftLeft128.c
|
||||
Common/SoftFloat-3e/s_approxRecip32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecip_1Ks.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt_1Ks.c
|
||||
@@ -75,12 +63,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/extF80_roundToInt.c
|
||||
Common/SoftFloat-3e/extF80_eq.c
|
||||
Common/SoftFloat-3e/extF80_lt.c
|
||||
Common/SoftFloat-3e/s_lt128.c
|
||||
Common/SoftFloat-3e/s_mul64ByShifted32To128.c
|
||||
Common/SoftFloat-3e/s_mul64To128.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros8.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros32.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros64.c
|
||||
Common/SoftFloat-3e/f32_to_extF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
@@ -88,6 +70,7 @@ set (SRCS
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
@@ -105,6 +88,7 @@ set (SRCS
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/StringCompareFallbacks.cpp
|
||||
Interface/Core/JIT/JIT.cpp
|
||||
Interface/Core/JIT/ALUOps.cpp
|
||||
Interface/Core/JIT/AtomicOps.cpp
|
||||
@@ -175,7 +159,7 @@ else()
|
||||
endif()
|
||||
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ=1;-DINLINE=static inline;-DINLINE_LEVEL=4;-DSOFTFLOAT_FAST_INT64=1;-DSOFTFLOAT_FAST_DIV32TO16=1;-DSOFTFLOAT_FAST_DIV64TO32=1")
|
||||
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
@@ -337,6 +321,10 @@ add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
target_link_libraries(FEXCore_Base TracyClient)
|
||||
endif()
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
|
||||
@@ -21,12 +21,12 @@ struct BitSet final {
|
||||
ElementType* Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
@@ -68,7 +68,7 @@ struct BitSetView final {
|
||||
ElementType* Memory;
|
||||
|
||||
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "Common/VectorRegType.h"
|
||||
|
||||
extern "C" {
|
||||
#include "SoftFloat-3e/platform.h"
|
||||
#include "SoftFloat-3e/softfloat.h"
|
||||
@@ -476,6 +478,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
FEXCore::VectorRegType Ret {};
|
||||
memcpy(&Ret, this, sizeof(*this));
|
||||
return Ret;
|
||||
}
|
||||
|
||||
LIBRARY_PRECISION ToFMax(softfloat_state* state) const {
|
||||
#ifdef _WIN32
|
||||
return ToF64(state);
|
||||
@@ -567,12 +575,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
*this = i32_to_extF80(rhs);
|
||||
}
|
||||
|
||||
X80SoftFloat(const FEXCore::VectorRegType rhs) {
|
||||
memcpy(this, &rhs, sizeof(*this));
|
||||
}
|
||||
|
||||
void operator=(extFloat80_t rhs) {
|
||||
Significand = rhs.signif;
|
||||
Exponent = rhs.signExp & 0x7FFF;
|
||||
Sign = rhs.signExp >> 15;
|
||||
}
|
||||
|
||||
operator FEXCore::VectorRegType() const {
|
||||
return ToVector();
|
||||
}
|
||||
|
||||
operator extFloat80_t() const {
|
||||
extFloat80_t Result {};
|
||||
Result.signif = Significand;
|
||||
|
||||
@@ -19,12 +19,24 @@ static bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
@@ -42,6 +54,13 @@ static bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, T* Result) {
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
#ifdef _M_ARM_64
|
||||
// Can't use uint8x16_t directly from arm_neon.h here.
|
||||
// Overrides softfloat-3e's defines which causes problems.
|
||||
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
|
||||
#elif defined(_M_X86_64)
|
||||
using VectorRegType = __m128i;
|
||||
#endif
|
||||
} // namespace FEXCore
|
||||
@@ -19,12 +19,10 @@
|
||||
|
||||
#include <array>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
@@ -113,16 +111,9 @@ fextl::string GetApplicationConfig(const std::string_view Program, bool Global)
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, uint64_t Config) {}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, const fextl::string& Config) {}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context* CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer* Meta {};
|
||||
class MetaLayer;
|
||||
static FEXCore::Config::MetaLayer* Meta {};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN, FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
@@ -143,9 +134,39 @@ public:
|
||||
~MetaLayer() {}
|
||||
void Load();
|
||||
|
||||
template<typename T>
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
if (std::holds_alternative<T>(Value)) [[likely]] {
|
||||
return std::get<T>(Value);
|
||||
}
|
||||
|
||||
T ConvertedValue;
|
||||
if (std::holds_alternative<fextl::string>(Value)) {
|
||||
const auto& StrVal = std::get<fextl::string>(Value);
|
||||
if (FEXCore::StrConv::Conv(StrVal, &ConvertedValue)) {
|
||||
// Convert the value.
|
||||
OptionMap[Option].emplace<T>(ConvertedValue);
|
||||
return ConvertedValue;
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Couldn't Convert {} to specified type!", StrVal);
|
||||
}
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
@@ -161,7 +182,7 @@ void MetaLayer::Load() {
|
||||
}
|
||||
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value) {
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
@@ -173,7 +194,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const FEXCore::Config::LayerValue& Value) {
|
||||
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
@@ -189,7 +210,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(MetaEnvironment->second);
|
||||
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
@@ -197,7 +218,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
Erase(Option);
|
||||
for (auto& Val : LookupMap) {
|
||||
// Set will emplace multiple options in to its list
|
||||
Set(Option, Val.first + "=" + Val.second);
|
||||
AppendStrArrayValue(Option, Val.first + "=" + Val.second);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -205,7 +226,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
@@ -214,7 +236,7 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
Meta = dynamic_cast<MetaLayer*>(ConfigLayers.begin()->second.get());
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
@@ -322,7 +344,7 @@ void ReloadMetaLayer() {
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, const fextl::string& PathName) {
|
||||
const auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
FEXCore::Config::Set(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -331,12 +353,17 @@ void ReloadMetaLayer() {
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory(false) + "RootFS/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
const auto PathNameCopy = *PathName;
|
||||
for (auto Global : {true, false}) {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedRootFS = DirectoryFetchers(Global) + "RootFS/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -353,12 +380,17 @@ void ReloadMetaLayer() {
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory(false) + "ThunkConfigs/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
const auto PathNameCopy = *PathName;
|
||||
for (auto Global : {true, false}) {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedConfig = DirectoryFetchers(Global) + "ThunkConfigs/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -373,12 +405,12 @@ void ReloadMetaLayer() {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_DUMPIR);
|
||||
if (*PathName != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP) && Meta->GetConv<bool>(FEXCore::Config::CONFIG_SINGLESTEP).value_or(false)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
}
|
||||
@@ -392,7 +424,7 @@ bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
@@ -400,6 +432,11 @@ std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
return Meta->GetConv<T>(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
@@ -408,31 +445,14 @@ void Erase(ConfigOption Option) {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
return Result;
|
||||
} else {
|
||||
return Default;
|
||||
auto Value = FEXCore::Config::GetConv<T>(Option);
|
||||
if (Value) {
|
||||
return *Value;
|
||||
}
|
||||
|
||||
return Default;
|
||||
}
|
||||
|
||||
template<>
|
||||
@@ -472,12 +492,13 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
|
||||
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
|
||||
DefaultValues::Type::StringArrayType* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -3,7 +3,7 @@
|
||||
"CPU": {
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
@@ -59,6 +59,8 @@
|
||||
"DISABLEFLAGM": "disableflagm",
|
||||
"ENABLEFLAGM2": "enableflagm2",
|
||||
"DISABLEFLAGM2": "disableflagm2",
|
||||
"ENABLEFRINTTS": "enablefrintts",
|
||||
"DISABLEFRINTTS": "disablefrintts",
|
||||
"ENABLECRYPTO": "enablecrypto",
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
@@ -90,19 +92,6 @@
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"CPUID": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::CPUID::OFF",
|
||||
"Enums": {
|
||||
"ENABLESHA": "enablesha",
|
||||
"DISABLESHA": "disablesha"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features are exposed in CPUID.",
|
||||
"\toff: Default CPU features queried from CPU features",
|
||||
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -363,6 +352,14 @@
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
|
||||
]
|
||||
},
|
||||
"ProfileStats": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
@@ -425,6 +422,14 @@
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -472,6 +477,13 @@
|
||||
"Sleeps the process at startup for a duration of seconds.",
|
||||
"Useful if an application crashes too quickly to attach a debugger."
|
||||
]
|
||||
},
|
||||
"StartupSleepProcName": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
|
||||
@@ -88,8 +88,11 @@ public:
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
|
||||
|
||||
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
|
||||
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
|
||||
|
||||
void ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) override;
|
||||
@@ -183,6 +186,10 @@ public:
|
||||
|
||||
void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) override;
|
||||
|
||||
void AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) override;
|
||||
|
||||
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
@@ -245,8 +252,6 @@ public:
|
||||
~ContextImpl();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const BlockDelinkerFunc& delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
|
||||
@@ -268,7 +273,8 @@ public:
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
struct GenerateIRResult {
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
std::optional<IR::IRListView> IRView;
|
||||
IR::RegisterAllocationData* RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
@@ -279,15 +285,14 @@ public:
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
bool GeneratedIR;
|
||||
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]]
|
||||
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
IR::OpSize GetGPROpSize() const {
|
||||
return Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
@@ -333,6 +338,10 @@ protected:
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else if (Config.ParanoidTSO) {
|
||||
AtomicTSOEmulationEnabled = true;
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
@@ -353,8 +362,6 @@ private:
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
@@ -370,5 +377,7 @@ private:
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
|
||||
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
|
||||
fextl::set<uint64_t> ForceTSOInstructions;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -0,0 +1,158 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = IREmit->_Constant(A.Offset);
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->_Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
// For 64-bit AddrSize can be 32-bit or 64-bit
|
||||
// For 32-bit AddrSize can be 32-bit or 16-bit
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->_Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
};
|
||||
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
// Loadstore rules:
|
||||
// Non-TSO GPR:
|
||||
// * LDR/STR: [Reg]
|
||||
// * LDR/STR: [Reg + Reg, {Shift <AccessSize>}]
|
||||
// * Can't use with 32-bit
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * Imm must be smaller than 16k with 32-bit
|
||||
// * LDUR/STUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// TSO GPR:
|
||||
// * ARMv8.0:
|
||||
// LDAR/STLR: [Reg]
|
||||
// * FEAT_LRCPC:
|
||||
// LDAPR: [Reg]
|
||||
// * FEAT_LRCPC2:
|
||||
// LDAPUR/STLUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// Non-TSO Vector:
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * LDUR/STUR: [Reg + [-256,255]]
|
||||
//
|
||||
// TSO Vector:
|
||||
// * ARMv8.0:
|
||||
// Just DMB + previous
|
||||
// * FEAT_LRCPC3 (Unsupported by FEXCore currently):
|
||||
// LDAPUR/STLUR: [Reg + [-256,255]]
|
||||
|
||||
const auto AccessSizeAsImm = OpSizeToSize(AccessSize);
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->_Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
|
||||
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
}
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && GPRSizeMatchesAddrSize && Is32Bit;
|
||||
|
||||
if (!Is32Bit || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->_Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
|
||||
struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -1,7 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
@@ -25,6 +24,22 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// LLVM's preserve_all doc, this is used throughout this file and reproduced
|
||||
// here for reference:
|
||||
//
|
||||
// the callee preserve all general purpose registers,
|
||||
// except X0-X8 and X16-X18. Furthermore it also preserves lower 128 bits of
|
||||
// V8-V31 SIMD - floating point registers.
|
||||
//
|
||||
// Note that the call necessarily also clobbers x30, the link register (LR)
|
||||
// which is not considered general purpose.
|
||||
//
|
||||
// Meanwhile, for non-preserve_all, the AAPCS64 ABI says:
|
||||
//
|
||||
// A subroutine invocation must preserve the contents of the registers
|
||||
// r19-r29 and SP.
|
||||
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
@@ -51,6 +66,13 @@ namespace x64 {
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 8> RA = {
|
||||
// All these callee saved
|
||||
ARMEmitter::Reg::r20, ARMEmitter::Reg::r21, ARMEmitter::Reg::r22, ARMEmitter::Reg::r23,
|
||||
@@ -59,6 +81,14 @@ namespace x64 {
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<ARMEmitter::Register, 2> PreserveAll_Dynamic = {
|
||||
ARMEmitter::Reg::r18,
|
||||
ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 2> NotPreserved_Dynamic = PreserveAll_Dynamic;
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
@@ -66,6 +96,11 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
@@ -75,6 +110,10 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
#else
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r8,
|
||||
@@ -94,15 +133,27 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r20,
|
||||
ARMEmitter::Reg::r21,
|
||||
ARMEmitter::Reg::r22,
|
||||
// PF/AF must be last.
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3,
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r8,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> RA = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = RA;
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
@@ -112,18 +163,20 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> PreserveAll_SRAFPR = {
|
||||
ARMEmitter::VReg::v0, ARMEmitter::VReg::v1, ARMEmitter::VReg::v2, ARMEmitter::VReg::v3,
|
||||
ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
ARMEmitter::VReg::v18, ARMEmitter::VReg::v19, ARMEmitter::VReg::v20, ARMEmitter::VReg::v21, ARMEmitter::VReg::v22,
|
||||
ARMEmitter::VReg::v23, ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
#endif
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
constexpr std::array<ARMEmitter::VRegister, 0> PreserveAll_DynamicFPR = {
|
||||
// None
|
||||
};
|
||||
#endif
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
@@ -147,16 +200,6 @@ namespace x64 {
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<ARMEmitter::Register, 1> PreserveAll_Dynamic = {
|
||||
// Only LR needs to get saved.
|
||||
ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
@@ -165,13 +208,6 @@ namespace x64 {
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
@@ -232,6 +268,11 @@ namespace x32 {
|
||||
ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = {
|
||||
ARMEmitter::Reg::r12, ARMEmitter::Reg::r13, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// All are caller saved
|
||||
@@ -284,17 +325,7 @@ namespace x32 {
|
||||
constexpr std::array<ARMEmitter::Register, 3> PreserveAll_Dynamic = {ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = 0;
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
@@ -355,18 +386,16 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralRegistersNotPreserved = x64::NotPreserved_Dynamic;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
#endif
|
||||
} else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
PairRegisters = x32::RAPairs;
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralRegistersNotPreserved = x32::NotPreserved_Dynamic;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
@@ -610,7 +639,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
#endif
|
||||
|
||||
if (SetPredRegs) {
|
||||
if (SetPredRegs && (EmitterCTX->HostFeatures.SupportsSVE256 || EmitterCTX->HostFeatures.SupportsSVE128)) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
@@ -622,6 +651,9 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
// Fill in the predicate register for the x87 ldst SVE optimization.
|
||||
ptrue(ARMEmitter::SubRegSize::i16Bit, PRED_X87_SVEOPT, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -673,10 +705,12 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
|
||||
|
||||
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
stp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), PFOffset);
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
@@ -822,8 +856,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -927,9 +960,9 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const ARMEmitter::Register> Reg
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
void Arm64Emitter::PushDynamicRegs(ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = GeneralRegistersNotPreserved.size() * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE256 ? 32 : 16;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
@@ -945,25 +978,17 @@ void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
PushVectorRegisters(TmpReg, CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Push the general registers.
|
||||
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
#endif
|
||||
PushGeneralRegisters(TmpReg, GeneralRegistersNotPreserved);
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
void Arm64Emitter::PopDynamicRegs() {
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
|
||||
// Pop vectors first
|
||||
PopVectorRegisters(CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Pop GPRs second
|
||||
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
#endif
|
||||
PopGeneralRegisters(GeneralRegistersNotPreserved);
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
@@ -1046,7 +1071,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
}
|
||||
|
||||
// Fill the static registers.
|
||||
FillStaticRegs(true, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
FillStaticRegs(FPRs, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
|
||||
// Pop the vector registers.
|
||||
PopVectorRegisters(CanUseSVE256, DynamicFPRs);
|
||||
|
||||
@@ -18,10 +18,8 @@
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -48,6 +46,10 @@ constexpr auto REG_AF = ARMEmitter::Reg::r27;
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = ARMEmitter::VReg::v0;
|
||||
constexpr auto VTMP2 = ARMEmitter::VReg::v1;
|
||||
|
||||
// Predicate register for X87 SVE Optimization
|
||||
constexpr auto SVE_OPT_PRED = ARMEmitter::PReg::p2;
|
||||
|
||||
#else
|
||||
constexpr auto TMP1 = ARMEmitter::XReg::x10;
|
||||
constexpr auto TMP2 = ARMEmitter::XReg::x11;
|
||||
@@ -67,6 +69,9 @@ constexpr auto VTMP2 = ARMEmitter::VReg::v17;
|
||||
constexpr auto EC_CALL_CHECKER_PC_REG = ARMEmitter::XReg::x9;
|
||||
constexpr auto EC_ENTRY_CPUAREA_REG = ARMEmitter::XReg::x17;
|
||||
|
||||
// Predicate register for X87 SVE Optimization
|
||||
constexpr auto SVE_OPT_PRED = ARMEmitter::PReg::p2;
|
||||
|
||||
// These structures are not included in the standard Windows headers, define the offsets of members we care about for EC here.
|
||||
constexpr size_t TEB_CPU_AREA_OFFSET = 0x1788;
|
||||
constexpr size_t TEB_PEB_OFFSET = 0x60;
|
||||
@@ -74,8 +79,16 @@ constexpr size_t PEB_EC_CODE_BITMAP_OFFSET = 0x368;
|
||||
constexpr size_t CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET = 0x1;
|
||||
constexpr size_t CPU_AREA_EMULATOR_STACK_BASE_OFFSET = 0x8;
|
||||
constexpr size_t CPU_AREA_EMULATOR_DATA_OFFSET = 0x30;
|
||||
|
||||
constexpr uint64_t EC_CODE_BITMAP_MAX_ADDRESS = 1ULL << 47;
|
||||
#endif
|
||||
|
||||
// Will force one single instruction block to be generated first if set when entering the JIT filling SRA.
|
||||
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP1;
|
||||
|
||||
// Predicate to use in the X87 SVE optimization
|
||||
constexpr ARMEmitter::PRegister PRED_X87_SVEOPT = ARMEmitter::PReg::p2;
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
@@ -91,9 +104,9 @@ protected:
|
||||
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegistersNotPreserved {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
@@ -128,8 +141,8 @@ protected:
|
||||
void PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
|
||||
void PopGeneralRegisters(std::span<const ARMEmitter::Register> Regs);
|
||||
|
||||
void PushDynamicRegsAndLR(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegsAndLR();
|
||||
void PushDynamicRegs(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegs();
|
||||
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
@@ -137,12 +150,12 @@ protected:
|
||||
// Spills and fills SRA/Dynamic registers that are required for Arm64 `preserve_all` ABI.
|
||||
// This ABI changes most registers to be callee saved.
|
||||
// Caller Saved:
|
||||
// - X0-X8, X16-X18.
|
||||
// - X0-X8, X16-X18, X30.
|
||||
// - v0-v7
|
||||
// - For 256-bit SVE hosts: top 128-bits of v8-v31
|
||||
//
|
||||
// Callee Saved:
|
||||
// - X9-X15, X19-X31
|
||||
// - X9-X15, X19-X29, X31
|
||||
// - Low 128-bits of v8-v31
|
||||
void SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs = true);
|
||||
void FillForPreserveAllABICall(bool FPRs = true);
|
||||
@@ -152,7 +165,7 @@ protected:
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
} else {
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
PushDynamicRegs(TmpReg);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -160,7 +173,7 @@ protected:
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
} else {
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,6 +39,19 @@ namespace CPU {
|
||||
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
|
||||
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
|
||||
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32_UPPER
|
||||
{0x5F00'0000'5F00'0000ULL, 0x5F00'0000'5F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I64
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32_UPPER
|
||||
{0x43E0'0000'0000'0000ULL, 0x43E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I64
|
||||
{0x8000'0000'8000'0000ULL, 0x8000'0000'8000'0000ULL}, // NAMED_VECTOR_CVTMAX_I32
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_I64
|
||||
{0x0000'0000'0000'0000ULL, 0x0000'0000'0000'8000ULL}, // NAMED_VECTOR_F80_SIGN_MASK
|
||||
{0x5A82'7999'5A82'7999ULL, 0x5A82'7999'5A82'7999ULL}, // NAMED_VECTOR_SHA1RNDS_K0
|
||||
{0x6ED9'EBA1'6ED9'EBA1ULL, 0x6ED9'EBA1'6ED9'EBA1ULL}, // NAMED_VECTOR_SHA1RNDS_K1
|
||||
{0x8F1B'BCDC'8F1B'BCDCULL, 0x8F1B'BCDC'8F1B'BCDCULL}, // NAMED_VECTOR_SHA1RNDS_K2
|
||||
{0xCA62'C1D6'CA62'C1D6ULL, 0xCA62'C1D6'CA62'C1D6ULL}, // NAMED_VECTOR_SHA1RNDS_K3
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
@@ -285,7 +298,7 @@ namespace CPU {
|
||||
// Fill in telemetry values
|
||||
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
|
||||
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -364,7 +377,7 @@ namespace CPU {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
|
||||
@@ -80,9 +80,13 @@ namespace CPU {
|
||||
struct JITCodeTail {
|
||||
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
|
||||
size_t Size;
|
||||
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// The length of the guest code for this block.
|
||||
size_t GuestSize;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
@@ -92,23 +96,10 @@ namespace CPU {
|
||||
// Shared-code modification spin-loop futex.
|
||||
uint32_t SpinLockFutex;
|
||||
|
||||
uint32_t _Pad;
|
||||
};
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using 16-bit entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
struct JITRIPReconstructEntries {
|
||||
// The Host PC offset from the previous entry.
|
||||
uint16_t HostPCOffset;
|
||||
|
||||
// How much to offset the RIP from the previous entry.
|
||||
uint16_t GuestRIPOffset;
|
||||
uint8_t _Pad[3];
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -119,14 +110,17 @@ namespace CPU {
|
||||
*
|
||||
* This is a thread specific compilation unit since there is one CPUBackend per guest thread
|
||||
*
|
||||
* @param Size - The byte size of the guest code for this block
|
||||
* @param SingleInst - If this block represents a single guest instruction
|
||||
* @param IR - IR that maps to the IR for this RIP
|
||||
* @param DebugData - Debug data that is available for this IR indirectly
|
||||
* @param CheckTF - If EFLAGS.TF checks should be emitted at the start of the block
|
||||
*
|
||||
* @return Information about the compiled code block.
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) = 0;
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
|
||||
@@ -192,7 +192,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
@@ -396,25 +395,7 @@ void CPUIDEmu::SetupFeatures() {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(CPUIDFeatures, CPUID);
|
||||
if (!CPUIDFeatures()) {
|
||||
// Early exit if no features are overriden.
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features.FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features.FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
ENABLE_DISABLE_OPTION(SHA, SHA, SHA);
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
Features.SHA = CTX->HostFeatures.SupportsSHA;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
|
||||
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -112,31 +113,59 @@ ContextImpl::~ContextImpl() {
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
struct GetFrameBlockInfoResult {
|
||||
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
|
||||
const CPU::CPUBackend::JITCodeTail* InlineTail;
|
||||
};
|
||||
static GetFrameBlockInfoResult GetFrameBlockInfo(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader*>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries*>(
|
||||
Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
return {InlineHeader, InlineTail};
|
||||
}
|
||||
|
||||
return {InlineHeader, nullptr};
|
||||
}
|
||||
|
||||
bool ContextImpl::IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail && (Address + Size > InlineTail->RIP && Address < InlineTail->RIP + InlineTail->GuestSize);
|
||||
}
|
||||
|
||||
bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail && InlineTail->SingleInst;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto [InlineHeader, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
|
||||
if (InlineHeader) {
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) && HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
|
||||
auto RIPEntry =
|
||||
reinterpret_cast<const uint8_t*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto& RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
auto HostPCOffset = FEXCore::Utils::vl64::Decode(RIPEntry);
|
||||
RIPEntry += HostPCOffset.Size;
|
||||
auto GuestRIPOffset = FEXCore::Utils::vl64::Decode(RIPEntry);
|
||||
RIPEntry += GuestRIPOffset.Size;
|
||||
if (HostPC >= (StartingHostPC + HostPCOffset.Integer)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
StartingHostPC += HostPCOffset.Integer;
|
||||
StartingGuestRIP += GuestRIPOffset.Integer;
|
||||
} else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
@@ -150,7 +179,8 @@ uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* T
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) {
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs,
|
||||
uint64_t PSTATE) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
uint32_t EFLAGS {};
|
||||
|
||||
@@ -160,6 +190,7 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
case X86State::RFLAG_TF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
@@ -212,6 +243,9 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
uint8_t TFByte = Frame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
EFLAGS |= (TFByte & 1) << X86State::RFLAG_TF_RAW_LOC;
|
||||
|
||||
// DF is pretransformed, undo the transform from 1/-1 back to 0/1
|
||||
uint8_t DFByte = Frame->State.flags[X86State::RFLAG_DF_RAW_LOC];
|
||||
if (DFByte & 0x80) {
|
||||
@@ -346,7 +380,7 @@ bool ContextImpl::InitCore() {
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
#elif !defined(_M_ARM64EC)
|
||||
#elif !defined(_M_ARM_64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
@@ -366,7 +400,7 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
@@ -439,6 +473,7 @@ void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
Profiler::PostForkAction(Child);
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
@@ -462,14 +497,10 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
@@ -488,40 +519,6 @@ static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter*
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
|
||||
// IRStorageBase with fully owned memory
|
||||
struct IRListCopy : public IR::IRStorageBase {
|
||||
std::span<std::byte> IRData;
|
||||
std::span<std::byte> ListData;
|
||||
|
||||
// TODO: Consider defaulting to empty RAData instead?
|
||||
IR::RegisterAllocationData::UniquePtr RADataInternal;
|
||||
|
||||
IRListCopy(const IR::IRListView& view, IR::RegisterAllocationData::UniquePtr RAData)
|
||||
: RADataInternal(std::move(RAData)) {
|
||||
std::byte* Storage = reinterpret_cast<std::byte*>(FEXCore::Allocator::malloc(view.GetDataSize() + view.GetListSize()));
|
||||
|
||||
IRData = {Storage, Storage + view.GetDataSize()};
|
||||
ListData = {Storage + view.GetDataSize(), Storage + view.GetDataSize() + view.GetListSize()};
|
||||
memcpy(IRData.data(), (char*)view.GetData(), IRData.size());
|
||||
memcpy(ListData.data(), (char*)view.GetListData(), ListData.size());
|
||||
}
|
||||
|
||||
IRListCopy(const IRListCopy& other) = delete;
|
||||
IRListCopy(IRListCopy&& other) = delete;
|
||||
|
||||
~IRListCopy() {
|
||||
FEXCore::Allocator::free(IRData.data());
|
||||
}
|
||||
|
||||
const IR::RegisterAllocationData* RAData() override {
|
||||
return RADataInternal.get();
|
||||
}
|
||||
IR::IRListView GetIRView() override {
|
||||
return IR::IRListView {IRData.data(), ListData.data(), IRData.size(), ListData.size()};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
@@ -550,6 +547,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst,
|
||||
[Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
@@ -567,6 +565,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
|
||||
bool BlockInForceTSOValidRange = false;
|
||||
auto InstForceTSOIt = ForceTSOInstructions.end();
|
||||
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
|
||||
InstForceTSOIt = It;
|
||||
BlockInForceTSOValidRange = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
@@ -583,6 +591,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
const FEXCore::X86Tables::DecodedInst* DecodedInfo {nullptr};
|
||||
|
||||
@@ -605,7 +614,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->FlushRegisterCache(true);
|
||||
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
@@ -622,7 +631,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -634,35 +643,52 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
IR::ForceTSOMode ForceTSO =
|
||||
BlockInForceTSOValidRange ?
|
||||
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
|
||||
IR::ForceTSOMode::ForceDisabled) :
|
||||
IR::ForceTSOMode::NoOverride;
|
||||
Thread->OpDispatcher->SetForceTSO(ForceTSO);
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
} else {
|
||||
if (Thread->OpDispatcher->HasHandledLock() != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", InstAddress, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
|
||||
// Walk InstForceTSOIt forward past the handled instruction
|
||||
InstForceTSOIt =
|
||||
std::find_if(InstForceTSOIt, ForceTSOInstructions.end(), [&](auto Val) { return Val >= Block.Entry + BlockInstructionsLength; });
|
||||
}
|
||||
} else {
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
if (!BlockInstructionsLength) {
|
||||
// SMC can modify block contents and patch invalid instructions to valid ones inline.
|
||||
// End blocks upon encountering them and only emit an invalid opcode exception if there are no prior instructions in the block (that could have modified it to be valid).
|
||||
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
}
|
||||
|
||||
HadInvalidInst = true;
|
||||
}
|
||||
|
||||
const bool NeedsBlockEnd =
|
||||
(HadDispatchError && TotalInstructions > 0) || (Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock);
|
||||
const bool NeedsBlockEnd = (HadDispatchError && TotalInstructions > 0) ||
|
||||
(Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock) || HadInvalidInst;
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {nullptr, 0, 0, 0, 0};
|
||||
return {{}, nullptr, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -694,19 +720,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IREmitter);
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr;
|
||||
|
||||
// Debug
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP,
|
||||
Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
IRDumper(Thread, IREmitter, GuestRIP, RAData);
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
auto IRList = fextl::make_unique<IRListCopy>(IREmitter->ViewIR(), std::move(RAData));
|
||||
|
||||
IREmitter->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
.IR = std::move(IRList),
|
||||
.IRView = IREmitter->ViewIR(),
|
||||
.RAData = RAData,
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
@@ -723,9 +746,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = CompiledCode,
|
||||
.IR = nullptr, // No IR/RA data generated
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.GeneratedIR = false, // nullptr here ensures IR cache mechanisms won't run
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
};
|
||||
@@ -740,55 +761,39 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto IRFromAOT = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP);
|
||||
if (IRFromAOT) {
|
||||
// Setup pointers to internal structures
|
||||
IR = std::move(IRFromAOT->IR);
|
||||
DebugData = IRFromAOT->DebugData;
|
||||
StartAddr = IRFromAOT->StartAddr;
|
||||
Length = IRFromAOT->Length;
|
||||
}
|
||||
// Generate IR + Meta Info
|
||||
auto [IRView, RAData, TotalInstructions, TotalInstructionsLength, StartAddr, Length] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
return {nullptr, nullptr, 0, 0};
|
||||
}
|
||||
auto DebugData = fextl::make_unique<FEXCore::Core::DebugData>();
|
||||
|
||||
if (!IR) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
// If the trap flag is set we generate single instruction blocks that each check to generate a single step exception.
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IR = std::move(IRCopy);
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
}
|
||||
|
||||
if (!IR) {
|
||||
return {};
|
||||
}
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
auto IRView = IR->GetIRView();
|
||||
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), RAData, TFSet);
|
||||
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, &IRView, DebugData, IR->RAData()).BlockEntry,
|
||||
.IR = std::move(IR),
|
||||
.DebugData = DebugData,
|
||||
.GeneratedIR = true,
|
||||
.CompiledCode = CompiledCode.BlockEntry,
|
||||
.DebugData = std::move(DebugData),
|
||||
.StartAddr = StartAddr,
|
||||
.Length = Length,
|
||||
};
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedJITTime);
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
@@ -799,7 +804,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
auto [CodePtr, IR, DebugData, GeneratedIR, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
@@ -850,14 +855,32 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, std::move(IR), DebugData, GeneratedIR)) {
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr);
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileSingleStep");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
@@ -903,19 +926,10 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
@@ -937,11 +951,11 @@ ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandl
|
||||
}
|
||||
|
||||
void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) {
|
||||
LOGMAN_THROW_AA_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
LOGMAN_THROW_A_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_A_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
if (!Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_A_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_A_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}", Entrypoint, GuestThunkEntrypoint);
|
||||
@@ -978,6 +992,19 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
ForceTSOValidRanges.Remove({Address, Address + Size});
|
||||
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
|
||||
@@ -46,6 +46,8 @@ Dispatcher::~Dispatcher() {
|
||||
}
|
||||
|
||||
void Dispatcher::EmitDispatcher() {
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
auto RipReg = TMP3;
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
#endif
|
||||
@@ -61,8 +63,9 @@ void Dispatcher::EmitDispatcher() {
|
||||
// }
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_Sleep;
|
||||
ARMEmitter::ForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_CompileSingleStep;
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -81,6 +84,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
ARMEmitter::ForwardLabel CompileSingleStep;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
@@ -89,6 +93,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
|
||||
@@ -116,10 +124,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
auto RipReg = TMP3;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
@@ -204,37 +213,21 @@ void Dispatcher::EmitDispatcher() {
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Clobbers TMP1/2
|
||||
auto EmitSignalGuardedRegion = [&](auto Body) {
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
Body();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
@@ -250,17 +243,38 @@ void Dispatcher::EmitDispatcher() {
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
};
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
auto EmitECExitCheck = [&]() {
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::SingleUseForwardLabel l_NotECCode;
|
||||
ARMEmitter::ForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
@@ -277,56 +291,83 @@ void Dispatcher::EmitDispatcher() {
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
};
|
||||
#endif
|
||||
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
// Jump to the compiled block
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
{
|
||||
Bind(&CompileSingleStep);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP1, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
b(&LoopTop);
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileSingleStep);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP }
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
// Jump to the compiled block
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -464,7 +505,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -488,7 +529,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
@@ -505,8 +546,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMF.GetConvertedPointer());
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMFCompileBlock.GetConvertedPointer());
|
||||
Bind(&l_CompileSingleStep);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
|
||||
dc64(PMFCompileSingleStep.GetConvertedPointer());
|
||||
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -75,7 +75,7 @@ Decoder::~Decoder() {
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
@@ -87,7 +87,7 @@ uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
}
|
||||
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
@@ -220,7 +220,8 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
// DecodeInst->TableInfo may be null in the case of 3DNow! ModRM decoding.
|
||||
const bool IsIndexVector = DecodeInst->TableInfo && (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
uint8_t InvalidSIBIndex = 0b100; ///< SIB Index where there is no register encoding.
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
@@ -234,7 +235,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Base = MapModRMToReg(BaseREX, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
if (Displacement) {
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
@@ -281,10 +282,10 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
|
||||
"should have "
|
||||
"been decoded "
|
||||
"before this!");
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
|
||||
"should have "
|
||||
"been decoded "
|
||||
"before this!");
|
||||
|
||||
uint8_t DestSize {};
|
||||
const bool HasWideningDisplacement =
|
||||
@@ -403,7 +404,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
@@ -470,7 +471,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
size_t CurrentSrc = 0;
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
|
||||
@@ -495,7 +498,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
@@ -514,14 +517,14 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
@@ -544,8 +547,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
}
|
||||
@@ -562,7 +565,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
|
||||
// A normal instruction is the most likely.
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INST) [[likely]] {
|
||||
@@ -606,13 +609,13 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
uint16_t LocalOp = OPD(Info->Type, PrefixType, ModRM.reg);
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
#undef OPD
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM) {
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM && ModRM.mod == 0b11) {
|
||||
// Everything in this group is privileged instructions aside from XGETBV
|
||||
constexpr std::array<uint8_t, 8> RegToField = {
|
||||
255, 0, 1, 2, 255, 255, 255, 3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
|
||||
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -689,7 +692,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
}
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_USES_EVEX_OPS, 1);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
}
|
||||
@@ -911,12 +914,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID, "Destination GPR was invalid");
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void Decoder::BranchTargetInMultiblockRange() {
|
||||
@@ -928,6 +925,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
uint64_t TargetRIP = 0;
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
bool Conditional = true;
|
||||
const auto InstEnd = DecodeInst->PC + DecodeInst->InstSize;
|
||||
|
||||
switch (DecodeInst->OP) {
|
||||
case 0x70 ... 0x7F: // Conditional JUMP
|
||||
@@ -936,17 +934,17 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
|
||||
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
// Target offset is PC + InstSize + Literal
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
TargetRIP = InstEnd + DecodeInst->Src[0].Literal();
|
||||
break;
|
||||
}
|
||||
case 0xE9:
|
||||
case 0xEB: // Both are unconditional JMP instructions
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
TargetRIP = InstEnd + DecodeInst->Src[0].Literal();
|
||||
Conditional = false;
|
||||
break;
|
||||
case 0xE8: // Call - Immediate target, We don't want to inline calls
|
||||
if (ExternalBranches) {
|
||||
ExternalBranches->insert(DecodeInst->PC + DecodeInst->InstSize);
|
||||
ExternalBranches->insert(InstEnd);
|
||||
}
|
||||
[[fallthrough]];
|
||||
case 0xC2: // RET imm
|
||||
@@ -960,7 +958,9 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress;
|
||||
// Forbid cross-page branches to both avoid massive (range-wise) code blocks in highly fragmented code and trying to decode unmapped branch targets
|
||||
bool ValidMultiblockMember =
|
||||
TargetRIP >= SymbolMinAddress && TargetRIP < std::min(FEXCore::AlignUp(InstEnd, FEXCore::Utils::FEX_PAGE_SIZE), SymbolMaxAddress);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
@@ -973,15 +973,10 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
MaxCondBranchBackwards = std::min(MaxCondBranchBackwards, TargetRIP);
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
CurrentBlockTargets.insert(FallthroughRIP);
|
||||
}
|
||||
AddBranchTarget(InstEnd);
|
||||
}
|
||||
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
CurrentBlockTargets.insert(TargetRIP);
|
||||
}
|
||||
AddBranchTarget(TargetRIP);
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
ExternalBranches->insert(TargetRIP);
|
||||
@@ -989,11 +984,15 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
if (FinalInstruction) {
|
||||
bool Decoder::InstCanContinue() const {
|
||||
if (DecodeInst->PC + DecodeInst->InstSize == NextBlockStartAddress) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!(DecodeInst->TableInfo->Flags & (FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t TargetRIP = 0;
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
@@ -1017,6 +1016,59 @@ bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
return false;
|
||||
}
|
||||
|
||||
void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
if (VisitedBlocks.contains(Target)) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), Target,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != Target, "unexpected");
|
||||
|
||||
if (BlockSuccIt != BlockInfo.Blocks.begin()) {
|
||||
auto BlockIt = std::prev(BlockSuccIt);
|
||||
if (BlockIt->Entry + BlockIt->Size > Target) {
|
||||
uint64_t SplitIdx = 0;
|
||||
uint64_t SplitAddr = BlockIt->Entry;
|
||||
// Find the instruction boundary of the split
|
||||
for (; SplitIdx < BlockIt->NumInstructions && SplitAddr < Target; SplitIdx++) {
|
||||
SplitAddr += BlockIt->DecodedInstructions[SplitIdx].InstSize;
|
||||
}
|
||||
uint64_t SplitOffset = SplitAddr - BlockIt->Entry;
|
||||
|
||||
LOGMAN_THROW_A_FMT(SplitIdx != 0, "unexpected");
|
||||
|
||||
if (SplitAddr == Target) {
|
||||
// Split at the boundary
|
||||
DecodedBlocks SplitBlock {
|
||||
.Entry = SplitAddr,
|
||||
.Size = BlockIt->Size - SplitOffset,
|
||||
.NumInstructions = BlockIt->NumInstructions - SplitIdx,
|
||||
.DecodedInstructions = BlockIt->DecodedInstructions + SplitIdx,
|
||||
.HasInvalidInstruction = BlockIt->HasInvalidInstruction,
|
||||
};
|
||||
|
||||
BlockIt->Size = SplitOffset;
|
||||
BlockIt->NumInstructions = SplitIdx;
|
||||
|
||||
BlockInfo.Blocks.insert(BlockSuccIt, SplitBlock);
|
||||
} // else misaligned, leave as a branch out of the block
|
||||
|
||||
// If we split a block then the target has already been visited as part of that, if it was
|
||||
// misaligned the jump will just leave the multiblock, mark it as visited to avoid running
|
||||
// this code path again and just bail out early.
|
||||
VisitedBlocks.insert(Target);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
CurrentBlockTargets.insert(Target);
|
||||
if (Target >= DecodeInst->PC + DecodeInst->InstSize && Target < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = Target;
|
||||
}
|
||||
}
|
||||
|
||||
const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
@@ -1040,7 +1092,7 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
@@ -1078,30 +1130,61 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
bool FinalInstruction {false};
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
while (!FinalInstruction && !BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlockInfo.Blocks.emplace_back();
|
||||
DecodedBlocks& CurrentBlockDecoding = BlockInfo.Blocks.back();
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
|
||||
CurrentBlockDecoding.Entry = RIPToDecode;
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
auto BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
|
||||
uint64_t PCOffset = 0;
|
||||
uint64_t BlockNumberOfInstructions {};
|
||||
uint64_t BlockStartOffset = DecodedSize;
|
||||
bool EraseBlock = true; // Unset once the block contains an instruction
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpMinAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
|
||||
InstructionSize = 0;
|
||||
|
||||
auto OpMinPage = OpMinAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
if (!EntryBlock && OpMinPage == OpMaxPage && PeekByte(0) == 0 && PeekByte(1) == 0) [[unlikely]] {
|
||||
// End the multiblock early if we hit 2 consecutive null bytes (add [rax], al) in the same page with the
|
||||
// assumption we are most likely trying to explore garbage code.
|
||||
break;
|
||||
}
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
CodePages.insert(CurrentCodePage);
|
||||
@@ -1112,64 +1195,66 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
CodePages.insert(CurrentCodePage);
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(OpAddress);
|
||||
uint64_t OpEndAddress = OpAddress + DecodeInst->InstSize;
|
||||
|
||||
if (ErrorDuringDecoding) [[unlikely]] {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
BlockIt->HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
DecodeInst->InstSize = 0;
|
||||
}
|
||||
|
||||
if (!ErrorDuringDecoding) {
|
||||
} else {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
auto TableInfo = DecodedBuffer[BlockStartOffset + BlockNumberOfInstructions].TableInfo;
|
||||
auto TableInfo = DecodeInst->TableInfo;
|
||||
if (!TableInfo || !TableInfo->OpcodeDispatcher) {
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
BlockIt->HasInvalidInstruction = true;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, RIPToDecode + PCOffset + DecodeInst->InstSize);
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, OpAddress);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, OpEndAddress);
|
||||
|
||||
if (OpEndAddress > NextBlockStartAddress) {
|
||||
// This instruction would overlap with another so skip adding it to the multiblock
|
||||
break;
|
||||
}
|
||||
|
||||
EraseBlock = false; // Block contains at least one valid instruction, so unset erase
|
||||
++TotalInstructions;
|
||||
++BlockNumberOfInstructions;
|
||||
++DecodedSize;
|
||||
++BlockIt->NumInstructions;
|
||||
BlockIt->Size += DecodeInst->InstSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) [[unlikely]] {
|
||||
if (BlockIt->HasInvalidInstruction) [[unlikely]] {
|
||||
if (!EntryBlock) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
TotalInstructions -= CurrentBlockDecoding.NumInstructions;
|
||||
TotalInstructions -= BlockIt->NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
BlockNumberOfInstructions = 0;
|
||||
InstStream -= PCOffset;
|
||||
CurrentBlockTargets.clear();
|
||||
EraseBlock = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
bool CanContinue = false;
|
||||
if (!(DecodeInst->TableInfo->Flags & (FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
|
||||
// If this isn't a block ender then we can keep going regardless
|
||||
CanContinue = true;
|
||||
// Check if we need to end the entire multiblock
|
||||
FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (FinalInstruction) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (!InstCanContinue()) {
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
BranchTargetInMultiblockRange();
|
||||
|
||||
// Bypass branches if we can continue through them in some cases.
|
||||
CanContinue |= BranchTargetCanContinue(FinalInstruction);
|
||||
}
|
||||
|
||||
if (FinalInstruction || !CanContinue) {
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1177,29 +1262,22 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
InstStream += DecodeInst->InstSize;
|
||||
}
|
||||
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
// NOTE: BlockIt is only valid here in the EraseBlock case
|
||||
if (EraseBlock) {
|
||||
BlockInfo.Blocks.erase(BlockIt);
|
||||
} else {
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
}
|
||||
|
||||
CurrentBlockTargets.clear();
|
||||
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
HasBlocks.emplace(RIPToDecode);
|
||||
|
||||
// Copy over only the number of instructions we decoded
|
||||
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockInfo.TotalInstructionCount += BlockNumberOfInstructions;
|
||||
|
||||
EntryBlock = false;
|
||||
}
|
||||
|
||||
BlockInfo.TotalInstructionCount = TotalInstructions;
|
||||
|
||||
for (auto CodePage : CodePages) {
|
||||
AddContainedCodePage(PC, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
|
||||
// sort for better branching
|
||||
std::sort(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(),
|
||||
[](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
|
||||
return a.Entry < b.Entry;
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -22,6 +22,7 @@ public:
|
||||
// New Frontend decoding
|
||||
struct DecodedBlocks final {
|
||||
uint64_t Entry {};
|
||||
uint64_t Size {};
|
||||
uint64_t NumInstructions {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
|
||||
bool HasInvalidInstruction {};
|
||||
@@ -70,7 +71,9 @@ private:
|
||||
bool DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
bool BranchTargetCanContinue(bool FinalInstruction) const;
|
||||
bool InstCanContinue() const;
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
|
||||
uint8_t ReadByte();
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
@@ -87,10 +90,10 @@ private:
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
const uint8_t* InstStream;
|
||||
const uint8_t* InstStream {};
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize;
|
||||
uint8_t InstructionSize {};
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
@@ -102,11 +105,12 @@ private:
|
||||
uint64_t SymbolMaxAddress {};
|
||||
uint64_t SymbolMinAddress {~0ULL};
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
uint64_t NextBlockStartAddress {~0ULL};
|
||||
|
||||
DecodedBlockInformation BlockInfo;
|
||||
fextl::set<uint64_t> CurrentBlockTargets;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t> VisitedBlocks;
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
@@ -120,7 +124,5 @@ private:
|
||||
};
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -1,22 +1,28 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/SoftFloat-3e/softfloat.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
softfloat_state State {};
|
||||
State.detectTininess = softfloat_tininess_afterRounding;
|
||||
State.exceptionFlags = 0;
|
||||
State.roundingPrecision = 80;
|
||||
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
if (!Force80BitPrecision) {
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
}
|
||||
|
||||
auto RC = (FCW >> 10) & 3;
|
||||
@@ -32,12 +38,14 @@ FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, float src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, float src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t FCW, double src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
@@ -45,7 +53,8 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
|
||||
bool eq, lt, nan;
|
||||
@@ -67,37 +76,43 @@ struct OpHandlers<IR::OP_F80CMP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF32(&State);
|
||||
return X80SoftFloat(src).ToF32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF64(&State);
|
||||
return X80SoftFloat(src).ToF64(&State);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI16(&State);
|
||||
return X80SoftFloat(src).ToI16(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI32(&State);
|
||||
return X80SoftFloat(src).ToI32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI64(&State);
|
||||
return X80SoftFloat(src).ToI64(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
auto rv = extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
auto rv = extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX || rv < INT16_MIN) {
|
||||
///< Indefinite value for 16-bit conversions.
|
||||
@@ -107,55 +122,63 @@ struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
return extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i64(&State, src, softfloat_round_minMag, false);
|
||||
return extF80_to_i64(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t FCW, int16_t src) {
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle2(uint16_t FCW, int16_t src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat(src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, int32_t src) {
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, int32_t src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FRNDINT(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::F2XM1(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FTAN(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSQRT(&State, Src1);
|
||||
}
|
||||
@@ -163,37 +186,42 @@ struct OpHandlers<IR::OP_F80SQRT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSIN(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FCOS(&State, Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FADD(&State, Src1, Src2);
|
||||
}
|
||||
@@ -201,7 +229,8 @@ struct OpHandlers<IR::OP_F80ADD> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSUB(&State, Src1, Src2);
|
||||
}
|
||||
@@ -209,7 +238,8 @@ struct OpHandlers<IR::OP_F80SUB> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FMUL(&State, Src1, Src2);
|
||||
}
|
||||
@@ -217,7 +247,8 @@ struct OpHandlers<IR::OP_F80MUL> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FDIV(&State, Src1, Src2);
|
||||
}
|
||||
@@ -225,103 +256,117 @@ struct OpHandlers<IR::OP_F80DIV> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FYL2X(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FATAN(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM1(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSCALE(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
}
|
||||
@@ -332,7 +377,9 @@ struct OpHandlers<IR::OP_F64SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1q, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
X80SoftFloat Src1 = Src1q;
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
@@ -373,7 +420,8 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
|
||||
uint64_t BCD {};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
@@ -15,13 +16,14 @@ static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHan
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex, false};
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, FEXCore::Core::CpuStateFrame*), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_PTR, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex, false};
|
||||
FallbackInfo
|
||||
GetFallbackInfo(double (*fn)(uint16_t, double, double, FEXCore::Core::CpuStateFrame*), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64_PTR, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
@@ -86,11 +88,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_F32_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_F64_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -100,11 +102,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -117,28 +119,31 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
*Info = {FABI_I16_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I16_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
*Info = {FABI_I32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
*Info = {FABI_I64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -147,7 +152,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
*Info = {FABI_I64_I16_F80_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
(Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP), SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -157,11 +162,13 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i16Bit: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_I16_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_I32_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -169,16 +176,16 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
@@ -229,7 +236,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I32_V128_V128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
|
||||
default: break;
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include <arm_neon.h>
|
||||
#endif
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#ifdef _M_ARM_64
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
if (is_using_words) {
|
||||
uint16x8_t a = vreinterpretq_u16_u8(data);
|
||||
uint16x8_t VIndexes {};
|
||||
const uint16x8_t VIndex16 = vdupq_n_u16(8);
|
||||
uint16_t Indexes[8] = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7,
|
||||
};
|
||||
memcpy(&VIndexes, Indexes, sizeof(VIndexes));
|
||||
auto MaskResult = vceqzq_u16(a);
|
||||
auto SelectResult = vbslq_u16(MaskResult, VIndexes, VIndex16);
|
||||
return vminvq_u16(SelectResult);
|
||||
} else {
|
||||
uint8x16_t VIndexes {};
|
||||
const uint8x16_t VIndex16 = vdupq_n_u8(16);
|
||||
uint8_t Indexes[16] = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
|
||||
};
|
||||
memcpy(&VIndexes, Indexes, sizeof(VIndexes));
|
||||
auto MaskResult = vceqzq_u8(data);
|
||||
auto SelectResult = vbslq_u8(MaskResult, VIndexes, VIndex16);
|
||||
return vminvq_u8(SelectResult);
|
||||
}
|
||||
}
|
||||
#else
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR uint32_t OpHandlers<IR::OP_VPCMPISTRX>::handle(FEXCore::VectorRegType lhs, FEXCore::VectorRegType rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
__uint128_t lhs_i;
|
||||
memcpy(&lhs_i, &lhs, sizeof(lhs_i));
|
||||
__uint128_t rhs_i;
|
||||
memcpy(&rhs_i, &rhs, sizeof(rhs_i));
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs_i, valid_lhs, rhs_i, valid_rhs, control);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
@@ -6,9 +7,9 @@
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Common/VectorRegType.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -344,51 +345,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(VectorRegType lhs, VectorRegType rhs, uint16_t control);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,8 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -16,22 +14,22 @@ struct IROp_Header;
|
||||
namespace FEXCore::CPU {
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_F80_I16_F32,
|
||||
FABI_F80_I16_F64,
|
||||
FABI_F80_I16_I16,
|
||||
FABI_F80_I16_I32,
|
||||
FABI_F32_I16_F80,
|
||||
FABI_F64_I16_F80,
|
||||
FABI_F64_I16_F64,
|
||||
FABI_F64_I16_F64_F64,
|
||||
FABI_I16_I16_F80,
|
||||
FABI_I32_I16_F80,
|
||||
FABI_I64_I16_F80,
|
||||
FABI_I64_I16_F80_F80,
|
||||
FABI_F80_I16_F80,
|
||||
FABI_F80_I16_F80_F80,
|
||||
FABI_F80_I16_F32_PTR,
|
||||
FABI_F80_I16_F64_PTR,
|
||||
FABI_F80_I16_I16_PTR,
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_F80_PTR,
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
|
||||
@@ -92,7 +92,7 @@ DEF_OP(AddNZCV) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmn(EmitSize, Src1, Const);
|
||||
} else if (IROp->Size < IR::OpSize::i32Bit) {
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
@@ -193,7 +193,7 @@ DEF_OP(TestNZ) {
|
||||
|
||||
DEF_OP(TestZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestZ>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < IR::OpSize::i32Bit, "TestNZ used at higher sizes");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size < IR::OpSize::i32Bit, "TestNZ used at higher sizes");
|
||||
const auto EmitSize = ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
@@ -202,7 +202,7 @@ DEF_OP(TestZ) {
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
|
||||
LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
LOGMAN_THROW_A_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
@@ -228,7 +228,7 @@ DEF_OP(SubNZCV) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
LOGMAN_THROW_A_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < IR::OpSize::i32Bit ? (32 - IR::OpSizeAsBits(OpSize)) : 0;
|
||||
@@ -287,7 +287,7 @@ DEF_OP(SetSmallNZV) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
setf8(GetReg(Op->Src.ID()).W());
|
||||
@@ -516,7 +516,7 @@ DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
@@ -536,7 +536,7 @@ DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
@@ -692,7 +692,7 @@ DEF_OP(ShiftFlags) {
|
||||
// updates for Src2=0 but anything that masks to zero.
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
@@ -773,7 +773,7 @@ DEF_OP(RotateFlags) {
|
||||
const auto EmitSize = Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
@@ -862,7 +862,7 @@ DEF_OP(PDep) {
|
||||
const auto T1 = TMP4.R();
|
||||
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
// First, copy the input/mask, since we'll be clobbering. Copy as 64-bit to
|
||||
// make this 0-uop on Firestorm.
|
||||
@@ -922,9 +922,9 @@ DEF_OP(PExt) {
|
||||
const auto BitReg = TMP2;
|
||||
const auto ValueReg = TMP3;
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel EarlyExit;
|
||||
ARMEmitter::ForwardLabel EarlyExit;
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
@@ -979,8 +979,8 @@ DEF_OP(LDiv) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check if the upper bits match the top bit of the lower 64-bits
|
||||
// Sign extend the top bit of lower bits
|
||||
@@ -1047,8 +1047,8 @@ DEF_OP(LUDiv) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
@@ -1115,8 +1115,8 @@ DEF_OP(LRem) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check if the upper bits match the top bit of the lower 64-bits
|
||||
// Sign extend the top bit of lower bits
|
||||
@@ -1187,8 +1187,8 @@ DEF_OP(LURem) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
@@ -1290,8 +1290,8 @@ DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1313,8 +1313,8 @@ DEF_OP(FindTrailingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeroes>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1338,8 +1338,8 @@ DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1360,8 +1360,8 @@ DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1386,7 +1386,10 @@ DEF_OP(Bfi) {
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
} else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
//
|
||||
// The move is 64-bit to allow register renaming, the upper bits don't
|
||||
// matter because of the bfi's EmitSize.
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, SrcDst);
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
} else {
|
||||
// Destination didn't match the dst source register.
|
||||
@@ -1428,8 +1431,8 @@ DEF_OP(Bfxil) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= IR::OpSize::i64Bit, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= IR::OpSize::i64Bit, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_A_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1438,7 +1441,7 @@ DEF_OP(Bfe) {
|
||||
if (Op->lsb == 0 && Op->Width == 32) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
} else if (Op->lsb == 0 && Op->Width == 64) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == IR::OpSize::i64Bit, "Must be 64-bit wide register");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == IR::OpSize::i64Bit, "Must be 64-bit wide register");
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
@@ -1549,13 +1552,13 @@ DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
[[maybe_unused]] constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1576,8 +1579,8 @@ DEF_OP(VExtractToGPR) {
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
LOGMAN_THROW_A_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_A_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
|
||||
// We need to use the upper 128-bit lane, so lets move it down.
|
||||
// Inverting our dedicated predicate for 128-bit operations selects
|
||||
|
||||
@@ -86,7 +86,7 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
size_t DataIndex {};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
|
||||
@@ -13,7 +13,7 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
LOGMAN_THROW_A_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
@@ -61,8 +61,8 @@ DEF_OP(CASPair) {
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
@@ -101,15 +101,20 @@ DEF_OP(CAS) {
|
||||
auto Expected = GetReg(Op->Expected.ID());
|
||||
auto Desired = GetReg(Op->Desired.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
if (Expected == Dst && Dst != MemSrc && Dst != Desired) {
|
||||
casal(SubEmitSize, Dst, Desired, MemSrc);
|
||||
} else {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
}
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
@@ -122,11 +127,11 @@ DEF_OP(CAS) {
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), Expected);
|
||||
mov(EmitSize, Dst, Expected);
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
@@ -274,7 +279,7 @@ DEF_OP(AtomicNeg) {
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
LOGMAN_THROW_A_FMT(
|
||||
OpSize == IR::OpSize::i64Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i8Bit, "Unexpecte"
|
||||
"d CAS "
|
||||
"size");
|
||||
@@ -286,7 +291,6 @@ DEF_OP(AtomicSwap) {
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
|
||||
@@ -53,14 +53,14 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
#ifdef _M_ARM_64EC
|
||||
if (RtlIsEcCode(NewRIP)) {
|
||||
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
#endif
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
ARMEmitter::ForwardLabel l_BranchHost;
|
||||
ldr(TMP1, &l_BranchHost);
|
||||
blr(TMP1);
|
||||
|
||||
@@ -72,7 +72,7 @@ DEF_OP(ExitFunction) {
|
||||
#endif
|
||||
} else {
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel FullLookup;
|
||||
ARMEmitter::ForwardLabel FullLookup;
|
||||
auto RipReg = GetReg(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
@@ -150,7 +150,7 @@ DEF_OP(Syscall) {
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
@@ -201,7 +201,7 @@ DEF_OP(Syscall) {
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
@@ -314,7 +314,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
|
||||
@@ -326,7 +326,7 @@ DEF_OP(Thunk) {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
FillStaticRegs(); // load from ctx after ra64 refill
|
||||
}
|
||||
@@ -378,7 +378,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// Arguments are passed as follows:
|
||||
@@ -397,7 +397,7 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
@@ -406,7 +406,7 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
@@ -433,7 +433,7 @@ DEF_OP(CPUID) {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be 4xi32 scalars
|
||||
@@ -446,7 +446,7 @@ DEF_OP(CPUID) {
|
||||
DEF_OP(XGetBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
@@ -467,7 +467,7 @@ DEF_OP(XGetBV) {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -16,7 +17,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
@@ -104,6 +105,16 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadTwoGPRs) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadTwoGPRs>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcLower = GetReg(Op->Lower.ID());
|
||||
const auto SrcUpper = GetReg(Op->Upper.ID());
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcLower);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcUpper, true);
|
||||
}
|
||||
|
||||
DEF_OP(VDupFromGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VDupFromGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -112,7 +123,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
@@ -206,7 +217,7 @@ DEF_OP(Vector_SToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -239,7 +250,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -270,7 +281,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
@@ -301,7 +312,7 @@ DEF_OP(Vector_FToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
@@ -404,7 +415,7 @@ DEF_OP(Vector_FToI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -459,13 +470,68 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToISized) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToISized>();
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = IROp->Size == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "256-bit not wired up, though we could change that");
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFRINTTS, "Need FRINTTS for Vector_FToISized");
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (ElementSize == IROp->Size) {
|
||||
// See above
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == IR::OpSize::i32Bit) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == IR::OpSize::i64Bit) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
if (Op->IntSize == IR::OpSize::i64Bit) {
|
||||
if (Op->HostRound) {
|
||||
ROUNDING_FN(frint64x);
|
||||
} else {
|
||||
ROUNDING_FN(frint64z);
|
||||
}
|
||||
} else {
|
||||
if (Op->HostRound) {
|
||||
ROUNDING_FN(frint32x);
|
||||
} else {
|
||||
ROUNDING_FN(frint32z);
|
||||
}
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
if (Op->IntSize == IR::OpSize::i64Bit) {
|
||||
if (Op->HostRound) {
|
||||
frint64x(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
} else {
|
||||
frint64z(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
} else {
|
||||
if (Op->HostRound) {
|
||||
frint32x(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
} else {
|
||||
frint32z(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_F64ToI32) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_F64ToI32>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -17,14 +17,14 @@ DEF_OP(VAESImc) {
|
||||
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -42,14 +42,14 @@ DEF_OP(VAESEnc) {
|
||||
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -65,14 +65,14 @@ DEF_OP(VAESEncLast) {
|
||||
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -90,14 +90,14 @@ DEF_OP(VAESDec) {
|
||||
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -169,6 +169,125 @@ DEF_OP(VSha1H) {
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
|
||||
DEF_OP(VSha1C) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1C>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1c(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1M) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1M>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1m(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1P) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1P>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1p(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1SU1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1SU1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1su1(Dst, Src2);
|
||||
} else if (Dst != Src2) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1su1(Dst, Src2);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1su1(VTMP1, Src2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h(Dst, Src2, Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha256h(Dst, Src2, Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256h(VTMP1, Src2, Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256H2) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H2>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256h2(VTMP1, Src2, Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
@@ -185,15 +304,32 @@ DEF_OP(VSha256U0) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
DEF_OP(VSha256U1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
if (Dst != Src1 && Dst != Src2) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
sha256su1(Dst, Src1, Src2);
|
||||
} else {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
sha256su1(VTMP1, Src1, Src2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
|
||||
@@ -11,6 +11,7 @@ desc: Main glue logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
@@ -21,6 +22,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
@@ -35,6 +37,7 @@ $end_info$
|
||||
#include <stdio.h>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
@@ -85,16 +88,13 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
auto FillF80Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
mov(TMP2, ARMEmitter::XReg::x1);
|
||||
mov(VTMP1.Q(), ARMEmitter::VReg::v0.Q());
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, TMP2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
};
|
||||
|
||||
auto FillF64Result = [&]() {
|
||||
@@ -118,52 +118,16 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
};
|
||||
|
||||
switch (Info.ABI) {
|
||||
case FABI_F80_I16_F32: {
|
||||
case FABI_F80_I16_F32_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, float, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
@@ -171,20 +135,59 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F32_I16_F80: {
|
||||
case FABI_F80_I16_F64_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_I16_PTR:
|
||||
case FABI_F80_I16_I32_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16_PTR) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x2, STATE);
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, uint32_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F32_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -196,43 +199,45 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
fmov(Dst.S(), VTMP1.S());
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F80: {
|
||||
case FABI_F64_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64: {
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
@@ -247,30 +252,31 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_I16_I16_F80: {
|
||||
case FABI_I16_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -281,38 +287,38 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I32_I16_F80: {
|
||||
case FABI_I32_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I64_I16_F80: {
|
||||
case FABI_I64_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -323,24 +329,28 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I64_I16_F80_F80: {
|
||||
case FABI_I64_I16_F80_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -351,42 +361,47 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_F80_I16_F80: {
|
||||
case FABI_F80_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
case FABI_F80_I16_F80_F80: {
|
||||
case FABI_F80_I16_F80_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint64_t>(
|
||||
ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
@@ -428,7 +443,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
case FABI_I32_V128_V128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
@@ -437,19 +452,22 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint16_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
@@ -464,12 +482,11 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(TMP1, &l_BranchHost);
|
||||
emit.blr(TMP1);
|
||||
emit.Bind(&l_BranchHost);
|
||||
@@ -484,11 +501,16 @@ static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
uintptr_t HostCode {};
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
if (!TFSet) {
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
}
|
||||
|
||||
if (!HostCode) {
|
||||
if (TFSet || !HostCode) {
|
||||
// If TF is set, the cache must be skipped as different code needs to be generated.
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
@@ -654,8 +676,69 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData,
|
||||
bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -711,22 +794,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
EmitInterruptChecks(CheckTF);
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -747,7 +815,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t*>();
|
||||
@@ -800,34 +868,61 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using two variable length integer entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
//
|
||||
// struct {
|
||||
// // The Host PC offset from the previous entry.
|
||||
// FEXCore::Utils::vl64 HostPCOffset;
|
||||
// // How much to offset the RIP from the previous entry.
|
||||
// FEXCore::Utils::vl64 GuestRIPOffset;
|
||||
// };
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
auto JITRIPEntriesBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->GuestSize = Size;
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
auto JITRIPEntriesLocation = JITRIPEntriesBegin;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto& GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto& RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
int64_t HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
int64_t GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
|
||||
size_t Size = FEXCore::Utils::vl64::Encode(JITRIPEntriesLocation, HostPCOffset);
|
||||
JITRIPEntriesLocation += Size;
|
||||
|
||||
Size = FEXCore::Utils::vl64::Encode(JITRIPEntriesLocation, GuestRIPOffset);
|
||||
JITRIPEntriesLocation += Size;
|
||||
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CursorIncrement(JITRIPEntriesLocation - JITRIPEntriesBegin);
|
||||
Align();
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
@@ -839,7 +934,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::STATS) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
LOGMAN_THROW_AA_FMT(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader");
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
LogMan::Msg::IFmt("RIP: 0x{:x}", Entry);
|
||||
LogMan::Msg::IFmt("Guest Code instructions: {}", HeaderOp->NumHostInstructions);
|
||||
|
||||
@@ -38,8 +38,9 @@ public:
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
CPUBackend::CompiledCode
|
||||
CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
@@ -56,10 +57,10 @@ private:
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::IR::IRListView* IR;
|
||||
uint64_t Entry;
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
const FEXCore::IR::IRListView* IR {};
|
||||
uint64_t Entry {};
|
||||
CPUBackend::CompiledCode CodeData {};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
@@ -68,7 +69,7 @@ private:
|
||||
ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
@@ -83,7 +84,7 @@ private:
|
||||
ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
@@ -110,7 +111,7 @@ private:
|
||||
ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Src, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "Only valid constant");
|
||||
return ARMEmitter::Reg::zr;
|
||||
} else {
|
||||
return GetReg(Src.ID());
|
||||
@@ -134,15 +135,15 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
return ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -157,7 +158,7 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
@@ -168,13 +169,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
@@ -185,13 +186,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
@@ -231,6 +232,10 @@ private:
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register ApplyMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, ARMEmitter::Register Tmp,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
@@ -252,9 +257,9 @@ private:
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass;
|
||||
const IR::RegisterAllocationData* RAData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
const IR::RegisterAllocationData* RAData {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
@@ -333,6 +338,9 @@ private:
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
|
||||
void EmitInterruptChecks(bool CheckTF);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
@@ -157,7 +158,7 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -209,7 +210,7 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -590,6 +591,44 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, ARMEmitter::Register Tmp,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
if (Offset.IsInvalid()) {
|
||||
return Base;
|
||||
}
|
||||
|
||||
if (OffsetScale != 1 && OffsetScale != IR::OpSizeToSize(AccessSize)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled OffsetScale: {}", OffsetScale);
|
||||
}
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Offset, &Const)) {
|
||||
if (Const == 0) {
|
||||
return Base;
|
||||
}
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const);
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
return Tmp;
|
||||
}
|
||||
|
||||
ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, [[maybe_unused]] uint8_t OffsetScale) {
|
||||
if (Offset.IsInvalid()) {
|
||||
@@ -603,10 +642,10 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
}
|
||||
|
||||
const auto SignedConst = static_cast<int64_t>(Const);
|
||||
const auto SignedAVXSize = static_cast<int64_t>(Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
const auto SignedSVESize = static_cast<int64_t>(HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
|
||||
const auto IsCleanlyDivisible = (SignedConst % SignedAVXSize) == 0;
|
||||
const auto Index = SignedConst / SignedAVXSize;
|
||||
const auto IsCleanlyDivisible = (SignedConst % SignedSVESize) == 0;
|
||||
const auto Index = SignedConst / SignedSVESize;
|
||||
|
||||
// SVE's immediate variants of load stores are quite limited in terms
|
||||
// of immediate range. They also operate on a by-vector-length basis.
|
||||
@@ -720,7 +759,8 @@ DEF_OP(LoadMemTSO) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -797,7 +837,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -852,7 +892,15 @@ DEF_OP(VLoadVectorMasked) {
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, TempDst.Q(), 0);
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
uint64_t Const {};
|
||||
if (Op->Offset.IsInvalid()) {
|
||||
// Intentional no-op.
|
||||
} else if (IsInlineConstant(Op->Offset, &Const)) {
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, MemReg, Const);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Complex addressing requested and not supported!");
|
||||
}
|
||||
|
||||
const uint64_t ElementSizeInBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
@@ -860,7 +908,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -892,7 +940,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -944,7 +992,16 @@ DEF_OP(VStoreVectorMasked) {
|
||||
// Use VTMP1 as the temporary destination
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
|
||||
uint64_t Const {};
|
||||
if (Op->Offset.IsInvalid()) {
|
||||
// Intentional no-op.
|
||||
} else if (IsInlineConstant(Op->Offset, &Const)) {
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, MemReg, Const);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Complex addressing requested and not supported!");
|
||||
}
|
||||
|
||||
const uint64_t ElementSizeInBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
@@ -952,7 +1009,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -981,6 +1038,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
@@ -1036,7 +1094,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
}
|
||||
|
||||
for (size_t i = DataElementOffsetStart, IndexElement = IndexElementOffsetStart; i < NumDataElements; ++i, ++IndexElement) {
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Extract mask element
|
||||
PerformMove(ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
@@ -1110,7 +1168,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
/// - AddrBase also doesn't need to exist
|
||||
/// - If the instruction is using 64-bit vector indexing or 32-bit addresses where the top-bit isn't set then this is valid!
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
@@ -1275,10 +1333,10 @@ DEF_OP(VLoadVectorElement) {
|
||||
const auto DstSrc = GetVReg(Op->DstSrc.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_MSG_A_FMT("Unsupported 256-bit VLoadVectorElement");
|
||||
@@ -1312,10 +1370,10 @@ DEF_OP(VStoreVectorElement) {
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
@@ -1341,16 +1399,16 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemReg = GetReg(Op->Address.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
@@ -1508,7 +1566,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
case IR::OpSize::i16Bit: strh(Src, MemSrc); break;
|
||||
@@ -1551,14 +1609,76 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemX87SVEOptPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemX87SVEOptPredicate>();
|
||||
const auto Predicate = PRED_X87_SVEOPT;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "StoreMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
const auto RegData = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
st1h<ARMEmitter::SubRegSize::i16Bit>(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
st1w<ARMEmitter::SubRegSize::i32Bit>(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
st1d(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} element size: {}", __func__, IROp->ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemX87SVEOptPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemX87SVEOptPredicate>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Predicate = PRED_X87_SVEOPT;
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "LoadMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
ld1h<ARMEmitter::SubRegSize::i16Bit>(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
ld1w<ARMEmitter::SubRegSize::i32Bit>(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ld1d(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} element size: {}", __func__, IROp->ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetReg(Op->Value1.ID());
|
||||
const auto Src2 = GetReg(Op->Value2.ID());
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), Addr, Op->Offset); break;
|
||||
case IR::OpSize::i64Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), Addr, Op->Offset); break;
|
||||
@@ -1590,10 +1710,11 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -1610,7 +1731,7 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -1662,7 +1783,7 @@ DEF_OP(MemSet) {
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Value = GetZeroableReg(Op->Value);
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
@@ -1680,8 +1801,8 @@ DEF_OP(MemSet) {
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel BackwardImpl {};
|
||||
ARMEmitter::SingleUseForwardLabel Done {};
|
||||
ARMEmitter::ForwardLabel BackwardImpl {};
|
||||
ARMEmitter::ForwardLabel Done {};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
if (Op->Prefix.IsInvalid()) {
|
||||
@@ -1718,7 +1839,6 @@ DEF_OP(MemSet) {
|
||||
case 8: stlr(Value.X(), TMP2); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
|
||||
if (Size >= 0) {
|
||||
@@ -1824,7 +1944,7 @@ DEF_OP(MemSet) {
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
LOGMAN_THROW_A_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
EmitMemset(DirectionConstant);
|
||||
} else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
@@ -1873,8 +1993,8 @@ DEF_OP(MemCpy) {
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel BackwardImpl {};
|
||||
ARMEmitter::SingleUseForwardLabel Done {};
|
||||
ARMEmitter::ForwardLabel BackwardImpl {};
|
||||
ARMEmitter::ForwardLabel Done {};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
mov(TMP2, MemRegDest.X());
|
||||
@@ -1923,23 +2043,23 @@ DEF_OP(MemCpy) {
|
||||
ldaprb(TMP4.W(), TMP3);
|
||||
stlrb(TMP4.W(), TMP2);
|
||||
} else {
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: ldaprh(TMP4.W(), TMP3); break;
|
||||
case 4: ldapr(TMP4.W(), TMP3); break;
|
||||
case 8: ldapr(TMP4, TMP3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
case 4: stlr(TMP4.W(), TMP2); break;
|
||||
case 8: stlr(TMP4, TMP2); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
} else {
|
||||
if (OpSize == 1) {
|
||||
@@ -1947,23 +2067,23 @@ DEF_OP(MemCpy) {
|
||||
ldarb(TMP4.W(), TMP3);
|
||||
stlrb(TMP4.W(), TMP2);
|
||||
} else {
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: ldarh(TMP4.W(), TMP3); break;
|
||||
case 4: ldar(TMP4.W(), TMP3); break;
|
||||
case 8: ldar(TMP4, TMP3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
case 4: stlr(TMP4.W(), TMP2); break;
|
||||
case 8: stlr(TMP4, TMP2); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2101,7 +2221,7 @@ DEF_OP(MemCpy) {
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
LOGMAN_THROW_A_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
EmitMemcpy(DirectionConstant);
|
||||
} else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
@@ -2121,13 +2241,15 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -2144,6 +2266,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
@@ -2157,6 +2280,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
|
||||
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
|
||||
@@ -2166,6 +2290,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
ldarb(TMP1, MemReg);
|
||||
@@ -2204,13 +2329,15 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -2225,7 +2352,8 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
@@ -2236,6 +2364,8 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
@@ -2394,7 +2524,7 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
@@ -2436,7 +2566,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
@@ -148,7 +148,7 @@ DEF_OP(PushRoundingMode) {
|
||||
} else if (Op->RoundMode == 0) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
LOGMAN_THROW_A_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
|
||||
@@ -166,7 +166,7 @@ DEF_OP(PopRoundingMode) {
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
@@ -189,7 +189,7 @@ DEF_OP(Print) {
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
@@ -267,7 +267,7 @@ DEF_OP(RDRAND) {
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
wfe();
|
||||
yield();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -44,11 +44,11 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_AA_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
|
||||
@@ -90,7 +90,7 @@ public:
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
LOGMAN_THROW_A_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
|
||||
@@ -6,6 +6,7 @@ desc: Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
@@ -26,7 +27,6 @@ $end_info$
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
|
||||
@@ -444,7 +444,7 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
} else {
|
||||
switch (SegmentReg) {
|
||||
@@ -466,7 +466,7 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -517,6 +517,8 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
// Unset the 'active' bit in the packed TF, skipping the single step exception after this instruction
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_TF_RAW_LOC>(_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), _Constant(1)));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
@@ -750,7 +752,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
auto OP = Op->OP & 0xF;
|
||||
auto [Complex, SimpleCond] = DecodeNZCVCondition(OP);
|
||||
if (Complex) {
|
||||
LOGMAN_THROW_AA_FMT(OP == 0xA || OP == 0xB, "only PF left");
|
||||
LOGMAN_THROW_A_FMT(OP == 0xA || OP == 0xB, "only PF left");
|
||||
CondJump_ = CondJumpBit(LoadPFRaw(false, false), 0, OP == 0xB);
|
||||
} else {
|
||||
CondJump_ = CondJumpNZCV(SimpleCond);
|
||||
@@ -997,6 +999,7 @@ void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
|
||||
uint64_t Const;
|
||||
bool AlwaysNonnegative = false;
|
||||
@@ -1089,8 +1092,8 @@ void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// Load both the source and the destination
|
||||
if (Op->OP == 0x90 && GetSrcSize(Op) >= 4 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX &&
|
||||
Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
if (Op->OP == 0x90 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX && Op->Dest.IsGPR() &&
|
||||
Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// This is one heck of a sucky special case
|
||||
// If we are the 0x90 XCHG opcode (Meaning source is GPR RAX)
|
||||
// and destination register is ALSO RAX
|
||||
@@ -1100,6 +1103,14 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// But this would result in a zext on 64bit, which would ruin the no-op nature of the instruction
|
||||
// So x86-64 spec mandates this special case that even though it is a 32bit instruction and
|
||||
// is supposed to zext the result, it is a true no-op
|
||||
//
|
||||
// x86 spec text here:
|
||||
//
|
||||
// XCHG (E)AX, (E)AX (encoded instruction byte is 90H) is an alias for
|
||||
// NOP regardless of data size prefixes, including REX.W.
|
||||
//
|
||||
// Note that also includes 16-bit so we don't gate this on size. The
|
||||
// sequence (66 90) is a valid two-byte nop that we also ignore.
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) {
|
||||
// If this instruction has a REP prefix then this is architecturally
|
||||
// defined to be a `PAUSE` instruction. On older processors this ends up
|
||||
@@ -1419,14 +1430,15 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// Allow garbage on the Src if it will be ignored by the Lshr below
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32});
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
// Allow garbage on the shift, we're masking it anyway.
|
||||
Ref Shift = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op.
|
||||
if (Size == 64) {
|
||||
Shift = _And(OpSize::i64Bit, Shift, _InlineConstant(0x3F));
|
||||
@@ -1535,7 +1547,7 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
Ref ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp1 = _Lshr(OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
|
||||
|
||||
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
|
||||
@@ -1586,7 +1598,7 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
|
||||
const uint32_t Size = GetSrcBitSize(Op);
|
||||
const auto OpSize = Size == 64 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t UnmaskedConst;
|
||||
uint64_t UnmaskedConst {};
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op. But it's
|
||||
// equivalent to mask to the actual size of the op, that way we can bound
|
||||
@@ -2173,7 +2185,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
uint64_t SrcConst;
|
||||
uint64_t SrcConst = 0;
|
||||
bool IsSrcConst = IsValueConstant(WrapNode(Src), &SrcConst);
|
||||
SrcConst &= 0x1f;
|
||||
|
||||
@@ -2397,7 +2409,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
unsigned LshrSize = std::max<uint8_t>(IR::OpSizeToSize(OpSize::i32Bit), Size / 8);
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : _And(OpSize::i64Bit, Src, _Constant(Mask));
|
||||
|
||||
// OF/SF/ZF/AF/PF undefined.
|
||||
// OF/SF/AF/PF undefined. ZF must be preserved. We choose to preserve OF/SF
|
||||
// too since we just use an rmif to insert into CF directly. We could
|
||||
// optimize perhaps.
|
||||
//
|
||||
// Set CF before the action to save a move, except for complements where we
|
||||
// can reuse the invert.
|
||||
if (Action != BTAction::BTComplement) {
|
||||
@@ -2405,7 +2420,8 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect);
|
||||
}
|
||||
|
||||
SetCFDirect_InvalidateNZV(Value, ConstantShift, Value);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
switch (Action) {
|
||||
@@ -2438,7 +2454,9 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = Dest;
|
||||
}
|
||||
|
||||
SetCFInverted_InvalidateNZV(Value, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
CFInverted = true;
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
@@ -2470,7 +2488,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2485,7 +2503,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2500,7 +2518,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2655,7 +2673,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
|
||||
Ref Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Result;
|
||||
Ref Result {};
|
||||
|
||||
if (Size != OpSize::i64Bit) {
|
||||
Src1 = _Bfe(OpSize::i64Bit, SizeBits, 0, Src1);
|
||||
@@ -2703,6 +2721,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto SizeBits = IR::OpSizeAsBits(Size);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
|
||||
Ref MaskConst {};
|
||||
if (Size == OpSize::i64Bit) {
|
||||
@@ -2998,6 +3017,22 @@ void OpDispatchBuilder::SGDTOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, _Constant(GDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
|
||||
auto DestAddress = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
|
||||
// See SGDTOp, matches Linux in reported values
|
||||
uint64_t IDTAddress = 0xFFFFFE0000000000ULL;
|
||||
auto IDTStoreSize = OpSize::i64Bit;
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
// Mask off upper bits if 32-bit result.
|
||||
IDTAddress &= ~0U;
|
||||
IDTStoreSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
_StoreMemAutoTSO(GPRClass, OpSize::i16Bit, DestAddress, _Constant(0xfff));
|
||||
_StoreMemAutoTSO(GPRClass, IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, _Constant(IDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SMSWOp(OpcodeArgs) {
|
||||
const bool IsMemDst = DestIsMem(Op);
|
||||
|
||||
@@ -3294,7 +3329,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(BeforeLoop);
|
||||
StartNewBlock();
|
||||
|
||||
ForeachDirection([this, Op, Size, REPE](int PtrDir) {
|
||||
ForeachDirection([this, Op, Size, REPE](int32_t PtrDir) {
|
||||
IRPair<IROp_CondJump> InnerJump;
|
||||
auto JumpIntoLoop = Jump();
|
||||
|
||||
@@ -3327,11 +3362,11 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
|
||||
// If TailCounter != 0, compare sources.
|
||||
@@ -3393,7 +3428,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int PtrDir) {
|
||||
ForeachDirection([this, Op, Size](int32_t PtrDir) {
|
||||
// XXX: Theoretically LODS could be optimized to
|
||||
// RSI += {-}(RCX * Size)
|
||||
// RAX = [RSI - Size]
|
||||
@@ -3437,7 +3472,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
|
||||
// Jump back to the start, we have more work to do
|
||||
@@ -3477,7 +3512,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int Dir) {
|
||||
ForeachDirection([this, Op, Size](int32_t Dir) {
|
||||
bool REPE = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX;
|
||||
|
||||
auto JumpStart = Jump();
|
||||
@@ -3521,7 +3556,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, _Constant(Dir * IR::OpSizeToSize(Size)));
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, _Constant(Dir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
@@ -3610,16 +3645,16 @@ void OpDispatchBuilder::DIVOp(OpcodeArgs) {
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, UDivOp, URemOp);
|
||||
StoreGPRRegister(X86State::REG_RAX, ResultAX, OpSize::i16Bit);
|
||||
} else if (Size == OpSize::i16Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
auto UDivOp = _LUDiv(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LURem(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp, Size);
|
||||
} else if (Size == OpSize::i32Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
Ref UDivOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LUDiv(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
Ref URemOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LURem(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
@@ -3651,7 +3686,7 @@ void OpDispatchBuilder::IDIVOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
if (Size == OpSize::i8Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, OpSize::i16Bit);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Src1 = _Sbfe(OpSize::i64Bit, 16, 0, Src1);
|
||||
Divisor = _Sbfe(OpSize::i64Bit, 8, 0, Divisor);
|
||||
|
||||
@@ -3662,16 +3697,16 @@ void OpDispatchBuilder::IDIVOp(OpcodeArgs) {
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, UDivOp, URemOp);
|
||||
StoreGPRRegister(X86State::REG_RAX, ResultAX, OpSize::i16Bit);
|
||||
} else if (Size == OpSize::i16Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
auto UDivOp = _LDiv(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LRem(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp, Size);
|
||||
} else if (Size == OpSize::i32Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
Ref UDivOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LDiv(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
Ref URemOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LRem(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
@@ -3769,7 +3804,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src1Lower = _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1);
|
||||
Src1Lower = Trivial ? Src1 : _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1);
|
||||
} else {
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, Size, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src1Lower = Src1;
|
||||
@@ -3806,15 +3841,9 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
Ref Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
HandledLock = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
Ref Src3 {};
|
||||
Ref Src3Lower {};
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src3 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Src3Lower = _Bfe(OpSize::i32Bit, 32, 0, Src3);
|
||||
} else {
|
||||
Src3 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Src3Lower = Src3;
|
||||
}
|
||||
auto Src3 = LoadGPRRegister(X86State::REG_RAX);
|
||||
auto Src3Lower = _Bfe(OpSize::i64Bit, OpSizeAsBits(Size), 0, Src3);
|
||||
|
||||
// If this is a memory location then we want the pointer to it
|
||||
Ref Src1 = MakeSegmentAddress(Op, Op->Dest);
|
||||
|
||||
@@ -3822,7 +3851,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
// Third operand must be a calculated guest memory address
|
||||
Ref CASResult = _CAS(Size, Src3Lower, Src2, Src1);
|
||||
Ref CASResult = _CAS(Size, Src3, Src2, Src1);
|
||||
Ref RAXResult = CASResult;
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src3Lower, CASResult);
|
||||
@@ -3921,7 +3950,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
Ref RealNode = reinterpret_cast<Ref>(GetNode(1));
|
||||
|
||||
[[maybe_unused]] const FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
LOGMAN_THROW_AA_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
|
||||
// Let's walk the jump blocks and see if we have handled every block target
|
||||
for (auto& Handler : JumpTargets) {
|
||||
@@ -3937,13 +3966,13 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
uint8_t OpDispatchBuilder::GetDstSize(X86Tables::DecodedOp Op) const {
|
||||
const uint32_t DstSizeFlag = X86Tables::DecodeFlags::GetSizeDstFlags(Op->Flags);
|
||||
LOGMAN_THROW_AA_FMT(DstSizeFlag != 0 && DstSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
LOGMAN_THROW_A_FMT(DstSizeFlag != 0 && DstSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
return 1u << (DstSizeFlag - 1);
|
||||
}
|
||||
|
||||
uint8_t OpDispatchBuilder::GetSrcSize(X86Tables::DecodedOp Op) const {
|
||||
const uint32_t SrcSizeFlag = X86Tables::DecodeFlags::GetSizeSrcFlags(Op->Flags);
|
||||
LOGMAN_THROW_AA_FMT(SrcSizeFlag != 0 && SrcSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
LOGMAN_THROW_A_FMT(SrcSizeFlag != 0 && SrcSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
return 1u << (SrcSizeFlag - 1);
|
||||
}
|
||||
|
||||
@@ -3993,7 +4022,7 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: return nullptr;
|
||||
}
|
||||
|
||||
CheckLegacySegmentRead(SegmentResult, Prefix);
|
||||
@@ -4123,94 +4152,6 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadEffectiveAddress(AddressMode A, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = _Constant(A.Offset);
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, Offset) : Offset;
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
LOGMAN_THROW_AA_FMT((A.IndexScale & (A.IndexScale - 1)) == 0, "power of two");
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
Tmp = _AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = _Lshl(GPRSize, A.Index, _Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
// For 64-bit AddrSize can be 32-bit or 64-bit
|
||||
// For 32-bit AddrSize can be 32-bit or 16-bit
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = _Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: _Constant(0);
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
// In the future this also needs to account for LRCPC3.
|
||||
bool SupportsRegIndex = Vector || !AtomicTSO;
|
||||
|
||||
// Try a constant offset. For 64-bit, this maps directly. For 32-bit, this
|
||||
// works only for displacements with magnitude < 16KB, since those bottom
|
||||
// addresses are reserved and therefore wrap around is invalid.
|
||||
//
|
||||
// TODO: Also handle GPR TSO if we can guarantee the constant inlines.
|
||||
if (SupportsRegIndex) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && A.AddrSize == OpSize::i32Bit && GPRSize == OpSize::i32Bit;
|
||||
|
||||
if ((A.AddrSize == OpSize::i64Bit) || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(B, true /* AddSegmentBase */, false),
|
||||
.Index = _Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Try a (possibly scaled) register index.
|
||||
if (A.AddrSize == OpSize::i64Bit && A.Base && (A.Index || A.Segment) && !A.Offset &&
|
||||
(A.IndexScale == 1 || A.IndexScale == IR::OpSizeToSize(AccessSize))) {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = _Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(A, true),
|
||||
.Index = InvalidNode,
|
||||
};
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
MemoryAccessType AccessType, bool IsLoad) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
@@ -4308,16 +4249,19 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
|
||||
if ((IsOperandMem(Operand, true) && LoadData) || ForceLoad) {
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemSrc = LoadEffectiveAddress(A, true);
|
||||
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, _Add(OpSize::i64Bit, MemSrc, _InlineConstant(8)));
|
||||
Ref MemSrc = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
return _LoadMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, MemSrc);
|
||||
} else {
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, _Add(OpSize::i64Bit, MemSrc, _InlineConstant(8)));
|
||||
}
|
||||
}
|
||||
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
} else {
|
||||
return LoadEffectiveAddress(A, false, AllowUpperGarbage);
|
||||
return LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), false, AllowUpperGarbage);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4416,9 +4360,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
Ref Value = GetOpSize(Src) == OpSize::i64Bit ? _Bfe(OpSize::i32Bit, 32, 0, Src) : Src;
|
||||
StoreGPRRegister(gpr, Value, GPRSize);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
LOGMAN_THROW_A_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(!(GPRSize == OpSize::i32Bit && OpSize > OpSize::i32Bit), "Oops had a {} GPR load", OpSize);
|
||||
LOGMAN_THROW_A_FMT(!(GPRSize == OpSize::i32Bit && OpSize > OpSize::i32Bit), "Oops had a {} GPR load", OpSize);
|
||||
|
||||
if (GPRSize != OpSize) {
|
||||
// if the GPR isn't the full size then we need to insert.
|
||||
@@ -4438,12 +4382,15 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A, true);
|
||||
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1);
|
||||
_StoreMem(GPRClass, OpSize::i16Bit, Upper, MemStoreDst, _Constant(8), std::min(Align, OpSize::i64Bit), MEM_OFFSET_SXTX, 1);
|
||||
Ref MemStoreDst = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, MemStoreDst);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1);
|
||||
_StoreMem(GPRClass, OpSize::i16Bit, Upper, MemStoreDst, _Constant(8), std::min(Align, OpSize::i64Bit), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
}
|
||||
@@ -4598,6 +4545,12 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LSLOp(OpcodeArgs) {
|
||||
// Emulate by always returning failure, this deviates from both Linux and Windows but
|
||||
// shouldn't be depended on by anything.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
IR::BreakDefinition Reason;
|
||||
bool SetRIPToNext = false;
|
||||
@@ -4877,12 +4830,13 @@ void OpDispatchBuilder::BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDe
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
Break(BreakDefinition);
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
if (Multiblock) {
|
||||
auto NextBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetCurrentCodeBlock(NextBlock);
|
||||
StartNewBlock();
|
||||
} else {
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4951,9 +4905,11 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
#define PF_3A_66 1
|
||||
constexpr static std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3A_AES[] = {
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
{OPD(1, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
};
|
||||
constexpr static std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3A_PCLMUL[] = {
|
||||
{OPD(0, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
{OPD(1, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
@@ -5077,9 +5033,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
@@ -5179,9 +5135,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::VPMULHWOp<false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::VPMULHWOp<true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
|
||||
@@ -5484,9 +5440,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
// 1 = Invalid
|
||||
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i32Bit>},
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i32Bit>},
|
||||
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i32Bit>},
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i32Bit>},
|
||||
|
||||
{OPDReg(0xD9, 4) | 0x00, 8, &OpDispatchBuilder::X87LDENVF64},
|
||||
|
||||
@@ -5581,7 +5537,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
// 6 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::f80Bit>},
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::f80Bit>},
|
||||
|
||||
|
||||
{OPD(0xDB, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
@@ -5632,9 +5588,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FISTF64, true>},
|
||||
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i64Bit>},
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i64Bit>},
|
||||
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i64Bit>},
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i64Bit>},
|
||||
|
||||
{OPDReg(0xDD, 4) | 0x00, 8, &OpDispatchBuilder::X87FRSTOR},
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
@@ -46,6 +47,12 @@ enum class BTAction {
|
||||
BTComplement,
|
||||
};
|
||||
|
||||
enum class ForceTSOMode {
|
||||
NoOverride,
|
||||
ForceDisabled,
|
||||
ForceEnabled,
|
||||
};
|
||||
|
||||
struct LoadSourceOptions {
|
||||
// Alignment of the load in bytes. iInvalid signifies opsize aligned.
|
||||
IR::OpSize Align = OpSize::iInvalid;
|
||||
@@ -72,19 +79,6 @@ struct LoadSourceOptions {
|
||||
bool AllowUpperGarbage = false;
|
||||
};
|
||||
|
||||
struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
@@ -273,6 +267,13 @@ public:
|
||||
return HandledLock;
|
||||
}
|
||||
|
||||
void SetForceTSO(ForceTSOMode Mode) {
|
||||
ForceTSO = Mode;
|
||||
}
|
||||
ForceTSOMode GetForceTSO() const {
|
||||
return ForceTSO;
|
||||
}
|
||||
|
||||
void SetDumpIR(bool DumpIR) {
|
||||
ShouldDump = DumpIR;
|
||||
}
|
||||
@@ -302,6 +303,7 @@ public:
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void LSLOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
void SyscallOp(OpcodeArgs, bool IsSyscallInst);
|
||||
void ThunkOp(OpcodeArgs);
|
||||
@@ -417,6 +419,7 @@ public:
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
void SIDTOp(OpcodeArgs);
|
||||
void SMSWOp(OpcodeArgs);
|
||||
|
||||
enum class VectorOpType {
|
||||
@@ -434,6 +437,7 @@ public:
|
||||
|
||||
void VectorALUROp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void RSqrt3DNowOp(OpcodeArgs, bool Duplicate);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
|
||||
@@ -466,10 +470,10 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void Scalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize, bool IsAVX);
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void MASKMOVOp(OpcodeArgs);
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType);
|
||||
@@ -515,12 +519,6 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVXScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void AVXVector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorScalarInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
@@ -715,32 +713,29 @@ public:
|
||||
RES_STI,
|
||||
};
|
||||
|
||||
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
|
||||
void FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FNINIT(OpcodeArgs);
|
||||
void FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FTST(OpcodeArgs);
|
||||
void FNINIT(OpcodeArgs);
|
||||
|
||||
void X87ModifySTP(OpcodeArgs, bool Inc);
|
||||
void X87SinCos(OpcodeArgs);
|
||||
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void FXCH(OpcodeArgs);
|
||||
void X87EMMS(OpcodeArgs);
|
||||
void X87FCMOV(OpcodeArgs);
|
||||
void X87FFREE(OpcodeArgs);
|
||||
void X87FLDCW(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs);
|
||||
void X87FNSAVE(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs);
|
||||
void X87FRSTOR(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87FXAM(OpcodeArgs);
|
||||
void X87FXTRACT(OpcodeArgs);
|
||||
void X87FCMOV(OpcodeArgs);
|
||||
void X87EMMS(OpcodeArgs);
|
||||
void X87FFREE(OpcodeArgs);
|
||||
|
||||
void FXCH(OpcodeArgs);
|
||||
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
void X87ModifySTP(OpcodeArgs, bool Inc);
|
||||
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
|
||||
|
||||
enum class FCOMIFlags {
|
||||
FLAGS_X87,
|
||||
@@ -749,39 +744,22 @@ public:
|
||||
void FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, FCOMIFlags WhichFlags, bool PopTwice);
|
||||
|
||||
// F64 X87 Ops
|
||||
void FLDF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FLDF64_Const(OpcodeArgs, uint64_t Num);
|
||||
|
||||
void FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FBLDF64(OpcodeArgs);
|
||||
void FBSTPF64(OpcodeArgs);
|
||||
|
||||
void FILDF64(OpcodeArgs);
|
||||
|
||||
void FSTF64(OpcodeArgs, IR::OpSize Width);
|
||||
|
||||
void FISTF64(OpcodeArgs, bool Truncate);
|
||||
|
||||
void FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FCOMIF64(OpcodeArgs, IR::OpSize width, bool Integer, FCOMIFlags whichflags, bool poptwice);
|
||||
void FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FILDF64(OpcodeArgs);
|
||||
void FISTF64(OpcodeArgs, bool Truncate);
|
||||
void FLDF64_Const(OpcodeArgs, uint64_t Num);
|
||||
void FLDF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FCHSF64(OpcodeArgs);
|
||||
void FABSF64(OpcodeArgs);
|
||||
void FTSTF64(OpcodeArgs);
|
||||
void FRNDINTF64(OpcodeArgs);
|
||||
void FSQRTF64(OpcodeArgs);
|
||||
void X87UnaryOpF64(OpcodeArgs, FEXCore::IR::IROps IROp);
|
||||
void X87BinaryOpF64(OpcodeArgs, FEXCore::IR::IROps IROp);
|
||||
void X87SinCosF64(OpcodeArgs);
|
||||
void X87FLDCWF64(OpcodeArgs);
|
||||
void X87TANF64(OpcodeArgs);
|
||||
void X87ATANF64(OpcodeArgs);
|
||||
void X87FXAMF64(OpcodeArgs);
|
||||
void X87FXTRACTF64(OpcodeArgs);
|
||||
void X87LDENVF64(OpcodeArgs);
|
||||
|
||||
void FCOMIF64(OpcodeArgs, IR::OpSize width, bool Integer, FCOMIFlags whichflags, bool poptwice);
|
||||
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
@@ -926,6 +904,15 @@ public:
|
||||
return Pair;
|
||||
}
|
||||
|
||||
Ref SHADataShuffle(Ref Src) {
|
||||
// SHA data shuffle matches PSHUFD shuffle where elements are inverted.
|
||||
// Because this shuffle mask gets reused multiple times per instruction, it's always a win to load the mask once and reuse it.
|
||||
const uint32_t Shuffle = 0b00'01'10'11;
|
||||
auto LookupIndexes =
|
||||
LoadAndCacheIndexedNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD, Shuffle * 16);
|
||||
return _VTBL1(OpSize::i128Bit, Src, LookupIndexes);
|
||||
}
|
||||
|
||||
RefPair AVX128_LoadSource_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
bool NeedsHigh, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
@@ -1029,7 +1016,7 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
@@ -1228,6 +1215,7 @@ public:
|
||||
uint64_t NextBit = (1ull << (Index - 1));
|
||||
uint32_t Offset = CacheIndexToContextOffset(Index);
|
||||
auto Class = CacheIndexClass(Index);
|
||||
LOGMAN_THROW_A_FMT(Offset != ~0U, "Invalid offset");
|
||||
|
||||
// Use stp where possible to store multiple values at a time. This accelerates AVX.
|
||||
// TODO: this is all really confusing because of backwards iteration,
|
||||
@@ -1345,6 +1333,7 @@ private:
|
||||
bool HandledLock {false};
|
||||
bool DecodeFailure {false};
|
||||
bool NeedsBlockEnd {false};
|
||||
ForceTSOMode ForceTSO {ForceTSOMode::NoOverride};
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
@@ -1468,7 +1457,10 @@ private:
|
||||
Ref Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
Ref Vector_CVT_Float_To_IntImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
Ref CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
|
||||
Ref Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf);
|
||||
|
||||
Ref Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen);
|
||||
|
||||
@@ -1502,9 +1494,6 @@ private:
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
|
||||
Ref LoadEffectiveAddress(AddressMode A, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize);
|
||||
|
||||
bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
// Literals are immediates as sources but memory addresses as destinations.
|
||||
return !(Load && Operand.IsLiteral()) && !Operand.IsGPR();
|
||||
@@ -1551,7 +1540,7 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
|
||||
}
|
||||
|
||||
@@ -1710,7 +1699,7 @@ private:
|
||||
CFInverted ^= true;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(CFInverted == RequiredInvert, "post condition");
|
||||
LOGMAN_THROW_A_FMT(CFInverted == RequiredInvert, "post condition");
|
||||
}
|
||||
|
||||
void CarryInvert() {
|
||||
@@ -1745,27 +1734,6 @@ private:
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
// As above but with
|
||||
//
|
||||
// x - 1
|
||||
//
|
||||
// If x = 0, hardware C is not set. If x = 1, hardware C is set.
|
||||
void SetCFInverted_InvalidateNZV(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
// This turns into a single rmif
|
||||
SetCFInverted(Value, ValueOffset, MustMask);
|
||||
} else {
|
||||
// Do math on flagm
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i64Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
HandleNZCVWrite();
|
||||
_SubNZCV(OpSize::i32Bit, Value, _InlineConstant(1));
|
||||
CFInverted = true;
|
||||
}
|
||||
}
|
||||
|
||||
void SetCFInverted(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
@@ -1788,6 +1756,13 @@ private:
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
|
||||
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
|
||||
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
|
||||
// unblocked bit before raising the exception.
|
||||
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, _Constant(0), _And(OpSize::i32Bit, PackedTF, _Constant(~1)), _Constant(1));
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
} else {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
@@ -1840,12 +1815,12 @@ private:
|
||||
static const int AVXHigh0Index = 48;
|
||||
static const int AVXHigh15Index = 63;
|
||||
|
||||
int CacheIndexToContextOffset(int Index) {
|
||||
uint32_t CacheIndexToContextOffset(int Index) {
|
||||
switch (Index) {
|
||||
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
|
||||
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
|
||||
default: return -1;
|
||||
default: return ~0U;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1885,7 +1860,7 @@ private:
|
||||
}
|
||||
|
||||
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index < 64, "valid index");
|
||||
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
|
||||
uint64_t Bit = (1ull << (uint64_t)Index);
|
||||
|
||||
if (Size == OpSize::i128Bit && (RegCache.Partial & Bit)) {
|
||||
@@ -1940,7 +1915,8 @@ private:
|
||||
}
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_A_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
|
||||
// Try to load a pair into the cache
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
@@ -1988,8 +1964,8 @@ private:
|
||||
}
|
||||
|
||||
void StoreContext(uint8_t Index, Ref Value) {
|
||||
LOGMAN_THROW_AA_FMT(Index < 64, "valid index");
|
||||
LOGMAN_THROW_AA_FMT(Value != InvalidNode, "storing valid");
|
||||
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
|
||||
LOGMAN_THROW_A_FMT(Value != InvalidNode, "storing valid");
|
||||
|
||||
uint64_t Bit = (1ull << (uint64_t)Index);
|
||||
|
||||
@@ -2381,11 +2357,15 @@ private:
|
||||
bool BlockSetRIP {false};
|
||||
|
||||
bool Multiblock {};
|
||||
uint64_t Entry;
|
||||
uint64_t Entry {};
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) {
|
||||
if (Class == FPRClass) {
|
||||
if (ForceTSO == ForceTSOMode::ForceEnabled) {
|
||||
return true;
|
||||
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
|
||||
return false;
|
||||
} else if (Class == FPRClass) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
@@ -2410,7 +2390,7 @@ private:
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
@@ -2420,6 +2400,7 @@ private:
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
AddressMode Out {};
|
||||
|
||||
@@ -2429,7 +2410,7 @@ private:
|
||||
A.Offset = 0;
|
||||
}
|
||||
|
||||
Out.Base = LoadEffectiveAddress(A, true, false);
|
||||
Out.Base = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true, false);
|
||||
return Out;
|
||||
}
|
||||
|
||||
@@ -2460,7 +2441,7 @@ private:
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
|
||||
@@ -116,8 +116,8 @@ void OpDispatchBuilder::InstallAVX128Handlers() {
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
@@ -217,9 +217,9 @@ void OpDispatchBuilder::InstallAVX128Handlers() {
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::AVX128_VPMULHW<false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::AVX128_VPMULHW<true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::AVX128_MOVVectorNT},
|
||||
|
||||
@@ -486,7 +486,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
|
||||
if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
LOGMAN_THROW_AA_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_A_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
return {
|
||||
.Low = AVX128_LoadXMMRegister(gprIndex, false),
|
||||
@@ -501,8 +501,8 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
HighA.Offset += 16;
|
||||
|
||||
if (Operand.IsSIB()) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_AA_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
[[maybe_unused]] const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
if (NeedsHigh) {
|
||||
@@ -523,10 +523,9 @@ OpDispatchBuilder::AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tabl
|
||||
|
||||
const auto Index_gpr = Operand.Data.SIB.Index;
|
||||
const auto Base_gpr = Operand.Data.SIB.Base;
|
||||
LOGMAN_THROW_AA_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
LOGMAN_THROW_A_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_A_FMT(Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
const auto Index_XMM_gpr = Index_gpr - X86State::REG_XMM_0;
|
||||
|
||||
return {
|
||||
@@ -542,7 +541,7 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
const RefPair Src, MemoryAccessType AccessType) {
|
||||
if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
LOGMAN_THROW_AA_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "expected AVX register");
|
||||
LOGMAN_THROW_A_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "expected AVX register");
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
if (Src.Low) {
|
||||
@@ -784,7 +783,7 @@ void OpDispatchBuilder::AVX128_VZERO(OpcodeArgs) {
|
||||
if (IsVZEROALL) {
|
||||
// NOTE: Despite the name being VZEROALL, this will still only ever
|
||||
// zero out up to the first 16 registers (even on AVX-512, where we have 32 registers)
|
||||
Ref ZeroVector;
|
||||
Ref ZeroVector {};
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
// Explicitly not caching named vector zero. This ensures that every register gets movi #0.0 directly.
|
||||
@@ -1058,18 +1057,8 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs) {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSizeFromSrc(Op), Op->Flags);
|
||||
}
|
||||
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if constexpr (HostRoundingMode) {
|
||||
Result = _Float_ToGPR_S(GPRSize, SrcElementSize, Src.Low);
|
||||
} else {
|
||||
Result = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src.Low);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
@@ -1337,10 +1326,12 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs) {
|
||||
};
|
||||
|
||||
Ref GPR {};
|
||||
if (SrcSize == OpSize::i128Bit && ElementSize == OpSize::i64Bit) {
|
||||
GPR = Mask8Byte(Src.Low);
|
||||
} else if (SrcSize == OpSize::i128Bit && ElementSize == OpSize::i32Bit) {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
if (Is128Bit) {
|
||||
if (ElementSize == OpSize::i64Bit) {
|
||||
GPR = Mask8Byte(Src.Low);
|
||||
} else {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
}
|
||||
} else if (ElementSize == OpSize::i32Bit) {
|
||||
auto GPRLow = Mask4Byte(Src.Low);
|
||||
auto GPRHigh = Mask4Byte(Src.High);
|
||||
@@ -1604,7 +1595,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
@@ -1614,46 +1605,20 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128BitSrc);
|
||||
RefPair Result {};
|
||||
|
||||
if (SrcElementSize == OpSize::i64Bit && Narrow) {
|
||||
///< Special case for VCVTPD2DQ/CVTTPD2DQ because it has weird rounding requirements.
|
||||
Result.Low = _Vector_F64ToI32(OpSize::i128Bit, Src.Low, HostRoundingMode ? Round_Host : Round_Towards_Zero, Is128BitSrc);
|
||||
|
||||
if (!Is128BitSrc) {
|
||||
// Also convert the upper 128-bit lane
|
||||
auto ResultHigh = _Vector_F64ToI32(OpSize::i128Bit, Src.High, HostRoundingMode ? Round_Host : Round_Towards_Zero, false);
|
||||
|
||||
// Zip the two halves together in to the lower 128-bits
|
||||
Result.Low = _VZip(OpSize::i128Bit, OpSize::i64Bit, Result.Low, ResultHigh);
|
||||
}
|
||||
} else {
|
||||
auto Convert = [this](Ref Src) -> Ref {
|
||||
auto ElementSize = SrcElementSize;
|
||||
if (Narrow) {
|
||||
ElementSize = ElementSize >> 1;
|
||||
Src = _Vector_FToF(OpSize::i128Bit, ElementSize, Src, SrcElementSize);
|
||||
}
|
||||
|
||||
if (HostRoundingMode) {
|
||||
return _Vector_FToS(OpSize::i128Bit, ElementSize, Src);
|
||||
} else {
|
||||
return _Vector_FToZS(OpSize::i128Bit, ElementSize, Src);
|
||||
}
|
||||
};
|
||||
|
||||
Result.Low = Convert(Src.Low);
|
||||
|
||||
if (!Is128BitSrc) {
|
||||
if (!Narrow) {
|
||||
Result.High = Convert(Src.High);
|
||||
} else {
|
||||
Result.Low = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, Result.Low, Convert(Src.High));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Narrow || Is128BitSrc) {
|
||||
Result.Low = Vector_CVT_Float_To_Int32Impl(Op, OpSize::i128Bit, Src.Low, OpSize::i128Bit, SrcElementSize, HostRoundingMode, Is128BitSrc);
|
||||
if (Is128BitSrc) {
|
||||
// Zero the upper 128-bit lane of the result.
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
} else {
|
||||
Result.High = Vector_CVT_Float_To_Int32Impl(Op, OpSize::i128Bit, Src.High, OpSize::i128Bit, SrcElementSize, HostRoundingMode, false);
|
||||
// Also convert the upper 128-bit lane
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
// Zip the two halves together in to the lower 128-bits
|
||||
Result.Low = _VZip(OpSize::i128Bit, OpSize::i64Bit, Result.Low, Result.High);
|
||||
|
||||
// Zero the upper 128-bit lane of the result.
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
}
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
@@ -1707,7 +1672,6 @@ void OpDispatchBuilder::AVX128_VEXTRACT128(OpcodeArgs) {
|
||||
const auto DstIsXMM = Op->Dest.IsGPR();
|
||||
const auto Selector = Op->Src[1].Literal() & 0b1;
|
||||
|
||||
///< TODO: Once we support loading only upper-half of the ymm register we can load the half depending on selection literal.
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
|
||||
|
||||
RefPair Result {};
|
||||
@@ -1853,7 +1817,7 @@ void OpDispatchBuilder::AVX128_VPERMQ(OpcodeArgs) {
|
||||
uint8_t SelectorLow = Selector & 0b1111;
|
||||
uint8_t SelectorHigh = (Selector >> 4) & 0b1111;
|
||||
auto SelectLane = [this](uint8_t Selector, RefPair Src) -> Ref {
|
||||
LOGMAN_THROW_AA_FMT(Selector < 16, "Selector too large!");
|
||||
LOGMAN_THROW_A_FMT(Selector < 16, "Selector too large!");
|
||||
|
||||
switch (Selector) {
|
||||
case 0b00'00: return _VDupElement(OpSize::i128Bit, OpSize::i64Bit, Src.Low, 0);
|
||||
@@ -2069,7 +2033,18 @@ void OpDispatchBuilder::AVX128_VPALIGNR(OpcodeArgs) {
|
||||
return Src2;
|
||||
}
|
||||
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src2, Index);
|
||||
if (Index == 16) {
|
||||
return Src1;
|
||||
}
|
||||
|
||||
auto SanitizedIndex = Index;
|
||||
if (Index > 16) {
|
||||
Src2 = Src1;
|
||||
Src1 = LoadZeroVector(OpSize::i128Bit);
|
||||
SanitizedIndex -= 16;
|
||||
}
|
||||
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src2, SanitizedIndex);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -2089,9 +2064,7 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
if (!Is128Bit) {
|
||||
///< TODO: This can be cleaner if AVX128_LoadSource_WithOpSize could return both constructed addresses.
|
||||
auto AddressHigh = _Add(OpSize::i64Bit, Address, _Constant(16));
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, AddressHigh, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
auto Address = MakeAddress(DataOp);
|
||||
@@ -2102,9 +2075,7 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
///< TODO: This can be cleaner if AVX128_LoadSource_WithOpSize could return both constructed addresses.
|
||||
auto AddressHigh = _Add(OpSize::i64Bit, Address, _Constant(16));
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, AddressHigh, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
@@ -62,29 +62,36 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
auto Src2 = SHADataShuffle(Src);
|
||||
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
// The result is swizzled differently than expected
|
||||
Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
} else {
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
}
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
@@ -92,16 +99,16 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1c?
|
||||
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p with different key
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1m
|
||||
return Self.BitwiseAtLeastTwo(B, C, D);
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
|
||||
@@ -119,60 +126,92 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
f3,
|
||||
};
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
Ref Result {};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Ref ConstantVector {};
|
||||
switch (Imm8) {
|
||||
case 0:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
|
||||
break;
|
||||
case 1:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
|
||||
break;
|
||||
case 2:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
|
||||
break;
|
||||
case 3:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
|
||||
break;
|
||||
}
|
||||
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Ref Src1 = SHADataShuffle(Dest);
|
||||
Ref Src2 = SHADataShuffle(Src);
|
||||
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
|
||||
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
switch (Imm8) {
|
||||
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
|
||||
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
|
||||
case 1:
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
} else {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
|
||||
StoreResult(FPRClass, Op, Dest0, OpSize::iInvalid);
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -222,19 +261,28 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
|
||||
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
Result = _VSha256U1(Src1, Src2);
|
||||
} else {
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
|
||||
StoreResult(FPRClass, Op, D0, OpSize::iInvalid);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
@@ -248,63 +296,88 @@ Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
|
||||
// Generates a suitable SHA256 `abcd` configuration from x86 format.
|
||||
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
};
|
||||
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
|
||||
// Generates a suitable SHA256 `efgh` configuration from x86 format.
|
||||
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
};
|
||||
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
auto ABCD = shuffle_abcd(Dest, Src);
|
||||
auto EFGH = shuffle_efgh(Dest, Src);
|
||||
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
|
||||
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
|
||||
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
auto A = _VSha256H(ABCD, EFGH, Key);
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
Result = shuffle_abcd(A, B);
|
||||
} else {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, OpSize::iInvalid);
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
|
||||
@@ -7,18 +7,18 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, true>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
|
||||
@@ -19,7 +19,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
FEXCore::X86State::RFLAG_CF_RAW_LOC, FEXCore::X86State::RFLAG_PF_RAW_LOC, FEXCore::X86State::RFLAG_AF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC, FEXCore::X86State::RFLAG_DF_RAW_LOC, FEXCore::X86State::RFLAG_OF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC, FEXCore::X86State::RFLAG_NT_LOC, FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC, FEXCore::X86State::RFLAG_AC_LOC, FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
@@ -135,6 +135,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
LOGMAN_THROW_A_FMT(SrcSize >= IR::OpSize::i8Bit && SrcSize <= IR::OpSize::i64Bit, "Invalid size");
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const uint64_t SignBit = IR::OpSizeAsBits(SrcSize) - 1;
|
||||
Ref Anded = nullptr;
|
||||
@@ -185,8 +186,9 @@ Ref OpDispatchBuilder::LoadAF() {
|
||||
// Read the result, stored for PF.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// What's left is to XOR and extract. This is the deferred part.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFWord, Result));
|
||||
// What's left is to XOR and extract. This is the deferred part. We
|
||||
// specifically use a 64-bit Xor here as we don't need masking.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i64Bit, AFWord, Result));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FixupAF() {
|
||||
@@ -199,7 +201,8 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
// Again 64-bit as masking is more expensive given our ConstProp design.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -238,8 +241,8 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
// appropriate bit. Again 64-bit to avoid masking.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
|
||||
@@ -6,40 +6,66 @@ namespace FEXCore::IR {
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
constexpr auto OpDispatchTableGenH0F3AREX = []<uint16_t REX>() consteval {
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> Table[] = {
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i8Bit>},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
|
||||
auto REX0 = OpDispatchTableGenH0F3AREX.template operator()<0>();
|
||||
auto REX1 = OpDispatchTableGenH0F3AREX.template operator()<1>();
|
||||
auto concat = []<typename T, size_t N1, size_t N2>(std::array<T, N1> const& lhs,
|
||||
std::array<T, N2> const& rhs) consteval -> std::array<T, N1 + N2> {
|
||||
std::array<T, N1 + N2> Table {};
|
||||
for (size_t i = 0; i < N1; ++i) {
|
||||
Table[i] = lhs[i];
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < N2; ++i) {
|
||||
Table[N1 + i] = rhs[i];
|
||||
}
|
||||
|
||||
return Table;
|
||||
};
|
||||
return concat(REX0, REX1);
|
||||
};
|
||||
|
||||
constexpr auto OpDispatch_H0F3ATableIgnoreREX = OpDispatchTableGenH0F3A();
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
|
||||
};
|
||||
|
||||
@@ -21,6 +21,11 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
@@ -36,6 +41,11 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x03, 1, &OpDispatchBuilder::LSLOp},
|
||||
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x0B, 1, &OpDispatchBuilder::INTOp},
|
||||
@@ -19,7 +20,6 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x32, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x34, 3, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
@@ -57,8 +57,8 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
@@ -143,8 +143,11 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
#endif
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepModTables[] = {
|
||||
@@ -161,7 +164,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
@@ -200,7 +203,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i64Bit>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
};
|
||||
|
||||
@@ -213,8 +216,8 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
|
||||
@@ -226,7 +229,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
@@ -284,7 +287,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
|
||||
@@ -626,17 +626,36 @@ void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i64Bit>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::RSqrt3DNowOp(OpcodeArgs, bool Duplicate) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto ElementSize = OpSize::i32Bit;
|
||||
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags);
|
||||
|
||||
// For the sqrt reciprocal in 3DNow!, if the source is negative,
|
||||
// then the result has the same sign as the source but the result is always calculated
|
||||
// as if the source was positive.
|
||||
Ref AbsSrc = _VFAbs(Size, ElementSize, Src);
|
||||
Ref PosRSqrt = _VFRSqrtPrecision(Size, ElementSize, AbsSrc);
|
||||
Ref Result = _VFCopySign(Size, ElementSize, PosRSqrt, Src);
|
||||
|
||||
if (Duplicate) {
|
||||
Result = _VDupElement(Size, ElementSize, Result, 0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
|
||||
// In the event of a scalar operation and a vector source, then
|
||||
// we can specify the entire vector length in order to avoid
|
||||
// unnecessary sign extension on the element to be operated on.
|
||||
// In the event of a memory operand, we load the exact element size.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(SrcSize, ElementSize, Src));
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags);
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(Size, ElementSize, Src));
|
||||
StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
@@ -676,8 +695,8 @@ void OpDispatchBuilder::VectorUnaryDuplicateOp(OpcodeArgs) {
|
||||
VectorUnaryDuplicateOpImpl(Op, IROp, ElementSize);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>(OpcodeArgs);
|
||||
// TODO: there's only one instantiation of this template. Lets remove it.
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVQOp(OpcodeArgs, VectorOpType VectorType) {
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
@@ -967,13 +986,17 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
// Special case element duplicate and broadcast to low or high 64-bits.
|
||||
return _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src, Shuffle & 0b11);
|
||||
}
|
||||
|
||||
case 0b00'00'10'10: {
|
||||
// Weird reverse low elements and broadcast to each half of the register
|
||||
Ref Tmp = _VUnZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
return _VZip(OpSize::i128Bit, OpSize::i32Bit, Tmp, Tmp);
|
||||
}
|
||||
case 0b00'00'11'10: {
|
||||
// First element duplicated and shifted in to the top.
|
||||
auto Dup = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dup, Src, 2);
|
||||
}
|
||||
case 0b00'01'00'01: {
|
||||
///< Weird reversed low elements and broadcast
|
||||
Ref Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
@@ -984,6 +1007,11 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
Ref Tmp = _VZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Tmp, Tmp, 4);
|
||||
}
|
||||
case 0b00'01'10'11: {
|
||||
// Inverse elements
|
||||
Ref Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i32Bit, Tmp, Tmp, 2);
|
||||
}
|
||||
case 0b00'10'00'10: {
|
||||
///< Weird reversed even elements and broadcast
|
||||
Ref Tmp = _VUnZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
@@ -1102,6 +1130,10 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
Ref Tmp = _VZip2(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Tmp, Tmp, 8);
|
||||
}
|
||||
case 0b10'11'00'01: {
|
||||
// Reverse each 64-bit lane.
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
}
|
||||
case 0b10'11'10'11: {
|
||||
///< Weird top two elements reverse and broadcast
|
||||
Ref Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src, Src);
|
||||
@@ -2067,6 +2099,35 @@ void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::AVXCVTGPR_To_FPR<OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXCVTGPR_To_FPR<OpSize::i64Bit>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFRINTTS) {
|
||||
// When we have FRINTTS, this is a two-step process. First, we round to the
|
||||
// right integer (where _Vector_FToISized matches x86 semantics), then just
|
||||
// convert that to a GPR.
|
||||
Src = _Vector_FToISized(SrcElementSize, SrcElementSize, Src, HostRoundingMode, GPRSize);
|
||||
return _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
} else {
|
||||
// When we lack hardware support, we need a bit of a convoluted sequence of
|
||||
// fixups before before and after conversion to emulate x86 semantics.
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
}
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
// If loading a vector, use the full size, so we don't
|
||||
@@ -2074,18 +2135,8 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if constexpr (HostRoundingMode) {
|
||||
Src = _Float_ToGPR_S(GPRSize, SrcElementSize, Src);
|
||||
} else {
|
||||
Src = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, GPRSize, OpSize::iInvalid);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
@@ -2127,77 +2178,57 @@ void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void OpDispatchBuilder::AVXVector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
Ref Result = Vector_CVT_Int_To_FloatImpl(Op, SrcElementSize, Widen);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_IntImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
auto ElementSize = SrcElementSize;
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
if (Narrow) {
|
||||
Src = _Vector_FToF(DstSize, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
ElementSize = ElementSize >> 1;
|
||||
}
|
||||
|
||||
if (HostRoundingMode) {
|
||||
return _Vector_FToS(DstSize, ElementSize, Src);
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf) {
|
||||
if (CTX->HostFeatures.SupportsFRINTTS && SrcSize != OpSize::i256Bit) {
|
||||
// If we have FRINTS, this is the usual 2-step
|
||||
Src = _Vector_FToISized(SrcSize, SrcElementSize, Src, HostRoundingMode, OpSize::i32Bit);
|
||||
Ref Dst = _Vector_FToZS(SrcSize, SrcElementSize, Src);
|
||||
if (SrcElementSize == OpSize::i32Bit) {
|
||||
// Return 32-bit result as-is
|
||||
return Dst;
|
||||
} else {
|
||||
// Down step from 64-bit ints to 32-bit ints
|
||||
return _VUShrNI(DstSize, SrcElementSize, Dst, 0);
|
||||
}
|
||||
} else {
|
||||
return _Vector_FToZS(DstSize, ElementSize, Src);
|
||||
// Otherwise, we have to do all the fixups, but vectorized.
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if (SrcElementSize == OpSize::i64Bit && Narrow) {
|
||||
///< Special case for CVTTPD2DQ because it has weird rounding requirements.
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result = _Vector_F64ToI32(DstSize, Src, HostRoundingMode ? Round_Host : Round_Towards_Zero, true);
|
||||
} else {
|
||||
Result = Vector_CVT_Float_To_IntImpl(Op, SrcElementSize, Narrow, HostRoundingMode);
|
||||
}
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, OpSizeFromSrc(Op), SrcElementSize, HostRoundingMode, true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::AVXVector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if (SrcElementSize == OpSize::i64Bit && Narrow) {
|
||||
///< Special case for CVTPD2DQ/CVTTPD2DQ because it has weird rounding requirements.
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result = _Vector_F64ToI32(DstSize, Src, HostRoundingMode ? Round_Host : Round_Towards_Zero, true);
|
||||
} else {
|
||||
Result = Vector_CVT_Float_To_IntImpl(Op, SrcElementSize, Narrow, HostRoundingMode);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, true>(OpcodeArgs);
|
||||
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op) {
|
||||
@@ -2277,7 +2308,7 @@ void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
// This function causes a change in MMX state from X87 to MMX
|
||||
if (MMXState == MMXState_X87) {
|
||||
@@ -2288,29 +2319,16 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
auto ElementSize = SrcElementSize;
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
if (Narrow) {
|
||||
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
ElementSize = ElementSize >> 1;
|
||||
}
|
||||
|
||||
if constexpr (HostRoundingMode) {
|
||||
Src = _Vector_FToS(Size, ElementSize, Src);
|
||||
} else {
|
||||
Src = _Vector_FToZS(Size, ElementSize, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, OpSize::iInvalid);
|
||||
Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, SrcSize, SrcElementSize, HostRoundingMode, false /* TODO? */);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
@@ -3956,6 +3974,7 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid size");
|
||||
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
|
||||
const auto MaskConstant = uint64_t {1} << (ElementSizeInBits - 1);
|
||||
|
||||
@@ -4657,8 +4676,7 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
if (Selector == 0xFF && Is256Bit) {
|
||||
Ref Result = Is256Bit ? Src2 : _VMov(OpSize::i128Bit, Src2);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResult(FPRClass, Op, Src2, OpSize::iInvalid);
|
||||
return;
|
||||
}
|
||||
// The only bits we care about from the 8-bit immediate for 128-bit operations
|
||||
@@ -4994,10 +5012,9 @@ OpDispatchBuilder::RefVSIB OpDispatchBuilder::LoadVSIB(const X86Tables::DecodedO
|
||||
|
||||
const auto Index_gpr = Operand.Data.SIB.Index;
|
||||
const auto Base_gpr = Operand.Data.SIB.Base;
|
||||
LOGMAN_THROW_AA_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
LOGMAN_THROW_A_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_A_FMT(Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
const auto Index_XMM_gpr = Index_gpr - X86State::REG_XMM_0;
|
||||
|
||||
return {
|
||||
@@ -5080,7 +5097,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
|
||||
// Only loads two 32-bit elements in to the lower 64-bits of the first destination.
|
||||
// Bits [255:65] all become zero.
|
||||
Result = _VMov(OpSize::i64Bit, Result);
|
||||
} else if (Is128Bit) {
|
||||
} else {
|
||||
Result = _VMov(OpSize::i128Bit, Result);
|
||||
}
|
||||
} else {
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -16,6 +17,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -39,7 +41,7 @@ Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW;
|
||||
Ref NewAbridgedFTW {};
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
Ref RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW);
|
||||
@@ -123,14 +125,17 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, shifted);
|
||||
ConvertedData = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, ConvertedData, _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, upper));
|
||||
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i128Bit, true, Width);
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale, /*Float=*/true);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
@@ -226,9 +231,9 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
const auto Result = (ResInST0 == OpResult::RES_STI) ? Offset : St0;
|
||||
const uint8_t Offset = Op->OP & 7;
|
||||
const uint8_t St0 = 0;
|
||||
const uint8_t Result = (ResInST0 == OpResult::RES_STI) ? Offset : St0;
|
||||
|
||||
if (Reverse ^ (ResInST0 == OpResult::RES_STI)) {
|
||||
_F80DivStack(Result, Offset, St0);
|
||||
@@ -534,8 +539,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, low);
|
||||
Mask = _VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, Mask, high);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
@@ -609,8 +613,8 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
uint8_t Offset = Op->OP & 7;
|
||||
Res = _F80CmpStack(Offset);
|
||||
} else {
|
||||
// Memory arg
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
@@ -618,6 +622,8 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
Res = _F80CmpValue(b);
|
||||
}
|
||||
@@ -749,13 +755,11 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FFREE(OpcodeArgs) {
|
||||
|
||||
_InvalidateStack(Op->OP & 7);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87EMMS(OpcodeArgs) {
|
||||
// Tags all get set to 0b11
|
||||
|
||||
_InvalidateStack(0xff);
|
||||
}
|
||||
|
||||
|
||||
@@ -103,15 +103,6 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i64Bit, true, Width);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
@@ -157,6 +148,8 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -193,6 +186,8 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -244,6 +239,8 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -299,6 +296,8 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -330,22 +329,22 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
// Implicit arg
|
||||
uint8_t offset = Op->OP & 7;
|
||||
b = _ReadStackValue(offset);
|
||||
} else {
|
||||
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
if (WhichFlags == FCOMIFlags::FLAGS_X87) {
|
||||
|
||||
@@ -145,7 +145,7 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
// These three are all X87 instructions
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9C, 1, X86InstInfo{"PUSHF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -21,49 +21,60 @@ constexpr uint16_t PF_3A_66 = 1;
|
||||
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
auto TableGen = []<uint16_t REX>() consteval {
|
||||
constexpr U16U8InfoStruct Table[] = {
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
constexpr auto H0F3ATable_IgnoresREX0 = TableGen.template operator()<0>();
|
||||
constexpr auto H0F3ATable_IgnoresREX1 = TableGen.template operator()<1>();
|
||||
|
||||
GenerateTable(&Table.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX0.at(0), H0F3ATable_IgnoresREX0.size());
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX1.at(0), H0F3ATable_IgnoresREX1.size());
|
||||
|
||||
constexpr U16U8InfoStruct TableNeedsREX[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
GenerateTable(&Table.at(0), TableNeedsREX, std::size(TableNeedsREX));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableIgnoreREX);
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableNeedsREX0);
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
static constexpr U16U8InfoStruct H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
|
||||
@@ -67,41 +67,41 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -23,7 +23,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0x01, 1, X86InstInfo{"", TYPE_GROUP_7, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
// These two load segment register data
|
||||
{0x02, 1, X86InstInfo{"LAR", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -260,6 +260,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
@@ -267,6 +268,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0, nullptr}},
|
||||
#endif
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
@@ -343,14 +343,14 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
|
||||
// x87
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
|
||||
|
||||
// Whether or not the instruction has a VEX prefix for the first source operand
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
|
||||
// Whether or not the instruction has a VEX prefix for the second source operand
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
|
||||
// Whether or not the instruction has a VEX prefix for the dest, first, or second source.
|
||||
constexpr InstFlagType FLAGS_VEX_SRC_MASK = (0b11ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_NO_OPERAND = (0b00ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (0b01ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (0b10ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (0b11ULL << 21);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 23);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -440,7 +440,7 @@ struct X86InstInfo {
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<X86InstInfo>::value, "X86InstInfo needs to be trivial");
|
||||
static_assert(std::is_trivially_copyable_v<X86InstInfo>);
|
||||
|
||||
constexpr size_t MAX_PRIMARY_TABLE_SIZE = 256;
|
||||
constexpr size_t MAX_SECOND_TABLE_SIZE = 256;
|
||||
|
||||
@@ -138,7 +138,7 @@ static bool LoadAOTIRCache(AOTIRCacheEntry* Entry, int streamfd) {
|
||||
|
||||
auto Array = (AOTIRInlineIndex*)((char*)FilePtr + IndexOffset);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
Entry->Array = Array;
|
||||
Entry->FilePtr = FilePtr;
|
||||
Entry->Size = Size;
|
||||
@@ -338,7 +338,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
const auto& FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
|
||||
// The lambda is converted to std::function. This is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
// NOTE: unique_ptr must be passed as a raw pointer since std::function requires lambda captures to be copyable
|
||||
@@ -368,10 +368,6 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
}
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
@@ -392,7 +388,7 @@ AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& fil
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry {.FileId = fileid, .Filename = filename}});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
|
||||
if (CTX->Config.AOTIRLoad && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
@@ -409,7 +405,7 @@ AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& fil
|
||||
|
||||
void AOTIRCaptureCache::UnloadAOTIRCacheEntry(AOTIRCacheEntry* Entry) {
|
||||
#ifndef _WIN32
|
||||
LOGMAN_THROW_AA_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
LOGMAN_THROW_A_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
|
||||
if (Entry->Array) {
|
||||
FEXCore::Allocator::munmap(Entry->FilePtr, Entry->Size);
|
||||
|
||||
@@ -159,7 +159,7 @@ struct NodeWrapperBase final {
|
||||
operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
|
||||
static_assert(std::is_trivially_copyable_v<NodeWrapperBase<OrderedNode>>);
|
||||
|
||||
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
|
||||
|
||||
@@ -355,7 +355,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_constructible_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
@@ -439,7 +439,7 @@ struct TypeDefinition final {
|
||||
operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
static_assert(std::is_trivially_copyable_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
using value_type = uint8_t;
|
||||
@@ -642,7 +642,9 @@ static inline OpSize operator/(IR::OpSize Size, T Divisor) {
|
||||
}
|
||||
|
||||
static inline uint8_t NumElements(IR::OpSize RegisterSize, IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid, "Invalid Size");
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid && RegisterSize != IR::OpSize::iUnsized &&
|
||||
ElementSize != IR::OpSize::iUnsized,
|
||||
"Invalid Size");
|
||||
return IR::OpSizeToSize(RegisterSize) / IR::OpSizeToSize(ElementSize);
|
||||
}
|
||||
|
||||
@@ -724,7 +726,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op);
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData);
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData);
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
|
||||
+288
-154
File diff suppressed because it is too large.
Load diff
@@ -37,7 +37,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, uint64_t Arg) {
|
||||
*out << "#0x" << std::hex << Arg;
|
||||
*out << "#0x" << std::hex << Arg << std::dec;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
@@ -82,7 +82,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData* RAData) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, const IR::RegisterAllocationData* RAData) {
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
const auto ArgID = Arg.ID();
|
||||
|
||||
@@ -206,6 +206,22 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
return "x87_log10_2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG_2:
|
||||
return "x87_log2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32:
|
||||
return "cvtmax_f32_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32_UPPER:
|
||||
return "cvtmax_f32_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I64:
|
||||
return "cvtmax_f32_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32:
|
||||
return "cvtmax_f64_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32_UPPER:
|
||||
return "cvtmax_f64_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I64:
|
||||
return "cvtmax_f64_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I32:
|
||||
return "cvtmax_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
|
||||
return "cvtmax_i64";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
}
|
||||
@@ -221,6 +237,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
case OpSize::i64Bit: *out << "i64"; break;
|
||||
case OpSize::i128Bit: *out << "i128"; break;
|
||||
case OpSize::i256Bit: *out << "i256"; break;
|
||||
case OpSize::f80Bit: *out << "f80"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
@@ -254,7 +271,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData) {
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
|
||||
@@ -160,7 +160,7 @@ IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(Ref insertA
|
||||
if (insertAfter) {
|
||||
LinkCodeBlocks(insertAfter, CodeNode);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(CurrentCodeBlock != nullptr, "CurrentCodeBlock must not be null here");
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBlock != nullptr, "CurrentCodeBlock must not be null here");
|
||||
|
||||
// Find last block
|
||||
auto LastBlock = CurrentCodeBlock;
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <new>
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
|
||||
@@ -61,6 +60,7 @@ public:
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
IRPair<IROp_Constant> _Constant(IR::OpSize Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
uint64_t Mask = ~0ULL >> (64 - IR::OpSizeAsBits(Size));
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size;
|
||||
@@ -206,7 +206,7 @@ public:
|
||||
|
||||
ReplaceAllUsesWithRange(Node, NewNode, Start, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Node->NumUses == 0, "Node still used");
|
||||
LOGMAN_THROW_A_FMT(Node->NumUses == 0, "Node still used");
|
||||
|
||||
auto IROp = Node->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_Header>();
|
||||
// We can not remove the op if there are side-effects
|
||||
@@ -357,10 +357,10 @@ protected:
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
|
||||
Ref InvalidNode;
|
||||
Ref InvalidNode {};
|
||||
Ref CurrentCodeBlock {};
|
||||
fextl::vector<Ref> CodeBlocks;
|
||||
uint64_t Entry;
|
||||
uint64_t Entry {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -98,11 +98,11 @@ protected:
|
||||
DualIntrusiveAllocator(size_t Size)
|
||||
: MemorySize {Size} {}
|
||||
|
||||
uintptr_t Data;
|
||||
uintptr_t List;
|
||||
uintptr_t Data {};
|
||||
uintptr_t List {};
|
||||
size_t DataCurrentOffset {0};
|
||||
size_t ListCurrentOffset {0};
|
||||
size_t MemorySize;
|
||||
size_t MemorySize {};
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorMalloc final : public DualIntrusiveAllocator {
|
||||
@@ -147,7 +147,6 @@ private:
|
||||
class IRListView final {
|
||||
public:
|
||||
IRListView() = delete;
|
||||
IRListView(IRListView&&) = delete;
|
||||
|
||||
IRListView(DualIntrusiveAllocator* Data)
|
||||
: IRListView(reinterpret_cast<void*>(Data->DataBegin()), reinterpret_cast<void*>(Data->ListBegin()), Data->DataSize(), Data->ListSize()) {}
|
||||
|
||||
@@ -70,7 +70,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass());
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->GetGPROpSize()));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
|
||||
@@ -80,7 +80,7 @@ public:
|
||||
void Finalize();
|
||||
|
||||
protected:
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
|
||||
private:
|
||||
using PassArrayType = fextl::vector<fextl::unique_ptr<Pass>>;
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
|
||||
namespace FEXCore {
|
||||
class CPUIDEmu;
|
||||
}
|
||||
struct HostFeatures;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class IntrusivePooledAllocator;
|
||||
@@ -19,7 +20,7 @@ class RegisterAllocationData;
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
|
||||
@@ -18,17 +18,13 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <string.h>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size >= IR::OpSize::i8Bit && Op->Size <= IR::OpSize::i64Bit, "Invalid mask size");
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
@@ -97,8 +93,8 @@ private:
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
}
|
||||
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IROp_Header* IROp, OrderedNodeWrapper Offset,
|
||||
MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IR::RegisterClassType RegisterClass, IROp_Header* IROp,
|
||||
OrderedNodeWrapper Offset, MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IREmit->IsValueConstant(Offset, &Imm)) {
|
||||
return;
|
||||
@@ -112,6 +108,9 @@ private:
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= (RegisterClass == GPRClass ? IR::OpSize::i64Bit : IR::OpSize::i256Bit),
|
||||
"Invalid "
|
||||
"size");
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(IROp->Size) - 1)) == 0 && Imm / IR::OpSizeToSize(IROp->Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
@@ -188,6 +187,35 @@ void ConstProp::HandleConstantPools(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
}
|
||||
|
||||
// Helper to replace the destination of an instruction with one of its sources,
|
||||
// to implement algebraic identities. This is surprisingly tricky due to
|
||||
// implicit masking in our IR.
|
||||
//
|
||||
// FEX's IR uses sized opcodes, matching arm64 semantics. 64-bit opcodes do not
|
||||
// mask, whereas smaller opcodes mask/zero-extend from 32-bits. Therefore, if
|
||||
// the instruction is 32-bit, we need to mask the source for a sound
|
||||
// replacement, in case there was garbage in the upper bits.
|
||||
//
|
||||
// However, if that source is in turn written by a 32-bit instruction, it is
|
||||
// guaranteed to have already been masked, so we know there's no garbage and we
|
||||
// can avoid the zero-extension. This is the case 99% of the time, but the
|
||||
// masking here is correctness-bearing nevertheless (and new versions of Denuvo
|
||||
// break if you get this wrong!)
|
||||
static inline void ReplaceWithSource(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Idx) {
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[Idx]);
|
||||
|
||||
if (IROp->Size < OpSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == OpSize::i32Bit, "other sizes not here");
|
||||
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[Idx]);
|
||||
if (Header->Size > OpSize::i32Bit) {
|
||||
Arg = IREmit->_Bfe(OpSize::i32Bit, 32, 0, Arg);
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
@@ -285,7 +313,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
Replaced = true;
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
Replaced = true;
|
||||
}
|
||||
|
||||
@@ -318,8 +346,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[1 - i]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 1 - i);
|
||||
Replaced = true;
|
||||
break;
|
||||
}
|
||||
@@ -361,8 +388,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
@@ -373,8 +399,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
@@ -398,6 +423,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
@@ -582,11 +608,41 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT:
|
||||
case OP_RMIFNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
break;
|
||||
}
|
||||
case OP_STORECONTEXT: {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (IROp->Size == OpSize::i128Bit) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
const auto MAX_STP_OFFSET = (252 * 4);
|
||||
|
||||
if (Op->Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Zero = IREmit->_Constant(0);
|
||||
Ref STP = IREmit->_StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Op->Offset);
|
||||
IREmit->Remove(CodeNode);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
Ref InlineZero = IREmit->_InlineConstant(0);
|
||||
IREmit->ReplaceNodeArgument(STP, 0, InlineZero);
|
||||
IREmit->ReplaceNodeArgument(STP, 1, InlineZero);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
@@ -640,27 +696,35 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, GPRClass, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMPAIR: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemPair>();
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value1_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value2_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
@@ -671,6 +735,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ private:
|
||||
};
|
||||
|
||||
IRDumper::IRDumper() {
|
||||
const auto DumpIRStr = DumpIR();
|
||||
const auto& DumpIRStr = DumpIR();
|
||||
if (DumpIRStr == "stderr" || DumpIRStr == "stdout" || DumpIRStr == "no") {
|
||||
// Intentionally do nothing
|
||||
} else if (DumpIRStr == "server") {
|
||||
@@ -53,7 +53,7 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
|
||||
auto IR = IREmit->ViewIR();
|
||||
auto HeaderOp = IR.GetHeader();
|
||||
LOGMAN_THROW_AA_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpToFile) {
|
||||
|
||||
@@ -65,7 +65,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
if (!EntryBlock) {
|
||||
EntryBlock = BlockNode;
|
||||
|
||||
Loaded 100 of 515 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user