mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 09:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7d3090f782 | ||
|
|
435dbdd014 | ||
|
|
d2b6e371e2 | ||
|
|
902d433a00 | ||
|
|
f11c8c747e | ||
|
|
47be6ad720 | ||
|
|
12326d7e71 | ||
|
|
3648ee97e8 | ||
|
|
e3f402a38a | ||
|
|
2aeff14e45 | ||
|
|
986c769f52 | ||
|
|
79a7afe7e1 | ||
|
|
7325cb09be | ||
|
|
c40675c6d0 | ||
|
|
1644d185c4 | ||
|
|
079a69dde2 | ||
|
|
c724742243 | ||
|
|
6788bcf264 | ||
|
|
6b43daccf0 | ||
|
|
67a7d4b352 | ||
|
|
aad2c969b7 | ||
|
|
0df84d3844 | ||
|
|
e320e1de79 | ||
|
|
63eb604b6c | ||
|
|
517a1836ee | ||
|
|
b1e54d803b | ||
|
|
4956ac23e5 | ||
|
|
59f85d6b7d | ||
|
|
d3dcc740b8 | ||
|
|
dd63548cb1 | ||
|
|
7962f40065 | ||
|
|
f3b0b41254 | ||
|
|
47557a6184 | ||
|
|
47bc73575e | ||
|
|
72b2ff88ae | ||
|
|
b37aae728f | ||
|
|
4e5c18113a | ||
|
|
ae3a4ae197 | ||
|
|
e2f973fe93 | ||
|
|
af8770005b | ||
|
|
0206479a08 | ||
|
|
0ddd46ba44 | ||
|
|
e2f0cf41d8 | ||
|
|
ff887adb19 | ||
|
|
638329278c | ||
|
|
08f451d3bc | ||
|
|
83d8317dae | ||
|
|
37d83d9375 | ||
|
|
2ea5cc3f41 | ||
|
|
4b418a81aa | ||
|
|
64a6da9aee | ||
|
|
0a41f470f8 | ||
|
|
229151f878 | ||
|
|
5006e02498 | ||
|
|
c747be4d8b | ||
|
|
e0c7419858 | ||
|
|
3aa38d12bc | ||
|
|
56a6e143ce | ||
|
|
cfafd8929a | ||
|
|
2005933518 | ||
|
|
177542e673 | ||
|
|
9c871c51c9 | ||
|
|
48d71752e2 | ||
|
|
82040bae3e | ||
|
|
555e8bd2d1 | ||
|
|
a0ba36eb1b | ||
|
|
69a3da4f31 | ||
|
|
ec502b99d8 | ||
|
|
c55fb3b5b8 | ||
|
|
1405755a31 | ||
|
|
06b7f3b232 | ||
|
|
7a82b07bfd | ||
|
|
54b178d151 | ||
|
|
26f0e5017f | ||
|
|
9243efc241 | ||
|
|
c91c6efc87 | ||
|
|
9bbdb35711 | ||
|
|
7723c47be4 | ||
|
|
eb41dac33f | ||
|
|
d119a5e7dc | ||
|
|
37eaff93cb | ||
|
|
ca1fb3f851 | ||
|
|
393f7d8702 | ||
|
|
19df181f65 | ||
|
|
d06e98ed84 | ||
|
|
2339831f6d | ||
|
|
93def42061 | ||
|
|
2790f89949 | ||
|
|
c3e9c3f70f | ||
|
|
4bc4d785af | ||
|
|
0309bc3d90 | ||
|
|
1523142540 | ||
|
|
5743282944 | ||
|
|
9d936a8989 | ||
|
|
1e95181b34 | ||
|
|
d85f5b4398 | ||
|
|
ed6d7092a2 | ||
|
|
a164315b7d | ||
|
|
e8084e3f11 | ||
|
|
17ebb38ebd | ||
|
|
fd7b749e47 | ||
|
|
1167d0354a | ||
|
|
6efbc2ba0b | ||
|
|
f2e8f580a4 | ||
|
|
5a24152672 | ||
|
|
fd208da77b | ||
|
|
cad4cb9b64 | ||
|
|
ea263166c5 | ||
|
|
3f66c6754d | ||
|
|
242f17eca9 | ||
|
|
7403789d8b | ||
|
|
ae84606da2 | ||
|
|
3f8f886f65 | ||
|
|
56a4f11ddb | ||
|
|
1695d742b1 | ||
|
|
5efc3212d2 | ||
|
|
82510eb452 | ||
|
|
7917f43720 | ||
|
|
58e0e1a03a | ||
|
|
8b469313f9 | ||
|
|
5f8a7a18f0 | ||
|
|
b3fe634c8b | ||
|
|
6f395885cb | ||
|
|
8cfbc44482 | ||
|
|
4731a5ace4 | ||
|
|
8201228c0d | ||
|
|
4170d9d50c | ||
|
|
3af19acdda | ||
|
|
cb9eb00339 | ||
|
|
98516b539e | ||
|
|
c18deb0d0f | ||
|
|
aa9ae27932 | ||
|
|
f9623af073 | ||
|
|
df9141ee7b | ||
|
|
5d6609a6b1 | ||
|
|
82e014cf5c | ||
|
|
4b652eb91b | ||
|
|
f18599d097 | ||
|
|
480308c021 | ||
|
|
89a13cd5fd | ||
|
|
794a11833d | ||
|
|
200f59ab70 | ||
|
|
91bbce5a96 | ||
|
|
af04940466 | ||
|
|
e431bfe0fa | ||
|
|
3397eb9b80 | ||
|
|
b5ac608fc5 | ||
|
|
8b01071e05 | ||
|
|
a9403db9a5 | ||
|
|
d96e48e246 | ||
|
|
86f363d202 | ||
|
|
208e9c31ea | ||
|
|
516ce6b15c | ||
|
|
7813c7aaa7 | ||
|
|
f32fe56bb3 | ||
|
|
27d8d57984 | ||
|
|
184c9522e1 | ||
|
|
f79de3c3f9 | ||
|
|
395b132f34 | ||
|
|
566a5e266d | ||
|
|
cc5686de37 | ||
|
|
049c932675 | ||
|
|
84fabf4ed8 | ||
|
|
310c2504f6 | ||
|
|
021c4fa4bf | ||
|
|
319c5bf023 | ||
|
|
023d0cdaa6 | ||
|
|
5a4f84cad9 | ||
|
|
eb27391954 | ||
|
|
ffaa5d6316 | ||
|
|
8bf22b6382 | ||
|
|
42c663269b | ||
|
|
983c5defc1 | ||
|
|
c92805475d | ||
|
|
50e6eee95a | ||
|
|
31cda95039 | ||
|
|
970ce4bc13 | ||
|
|
e4c6154be2 | ||
|
|
064a6e96c7 | ||
|
|
d704bd3cef | ||
|
|
9f511d5d78 | ||
|
|
a77d00acae | ||
|
|
5d11b8b0d5 | ||
|
|
bf70dee3ae | ||
|
|
4a065ad1a2 | ||
|
|
96d955e7b1 | ||
|
|
abec314720 | ||
|
|
ec63be4028 | ||
|
|
179b4423a7 | ||
|
|
9956a528b9 | ||
|
|
e3c6b37f70 | ||
|
|
b78bde7ec8 | ||
|
|
6ff7e82e3b | ||
|
|
37675e0351 | ||
|
|
38c9f14dbd | ||
|
|
7187bc0e59 | ||
|
|
c5700cb0cc | ||
|
|
8ed4c52d58 | ||
|
|
bd65c3e568 | ||
|
|
7052f91495 | ||
|
|
040d3115d2 | ||
|
|
7236c2bf0a | ||
|
|
ed3fa2f1c6 | ||
|
|
a454845ffd | ||
|
|
40bfff985f | ||
|
|
96d65cd6bc | ||
|
|
e4e45bcc59 | ||
|
|
45fbab212a | ||
|
|
8243b93e4d | ||
|
|
52cc5689db | ||
|
|
cfef3d2cbf | ||
|
|
fd64ea54fe | ||
|
|
d1b2333195 | ||
|
|
97fbd11177 | ||
|
|
e7836f26dd | ||
|
|
cf6da6ac26 | ||
|
|
a0cb91daa4 | ||
|
|
e010e8adf0 | ||
|
|
7fe7c39a00 | ||
|
|
b03dc847f5 | ||
|
|
ac6be77592 | ||
|
|
35330df728 | ||
|
|
bd9897b164 | ||
|
|
e704275ffd | ||
|
|
e33cb3c831 | ||
|
|
91264f05e2 | ||
|
|
30e4650a8b | ||
|
|
add92f681b | ||
|
|
21b8f8e20d | ||
|
|
ca055cd7d5 | ||
|
|
42e207932b | ||
|
|
3157685f0f | ||
|
|
d833f3ae67 | ||
|
|
fbf5187794 | ||
|
|
f2c8adf7a3 | ||
|
|
d472ce190c | ||
|
|
c4c82f23ee | ||
|
|
20e9e2044e | ||
|
|
1a0412969b | ||
|
|
55c80e6c4f | ||
|
|
83469208c9 | ||
|
|
3471ca67c4 | ||
|
|
4048faa54b | ||
|
|
61d94ac117 | ||
|
|
511c45c4c6 | ||
|
|
bce8b13366 | ||
|
|
4a7fda4516 | ||
|
|
b29561e131 | ||
|
|
5a68359f37 | ||
|
|
00b1249e53 | ||
|
|
8be7c1bf8e | ||
|
|
aaca8772bd | ||
|
|
ab6e67596f | ||
|
|
433091c326 | ||
|
|
c01bfcd52f | ||
|
|
673a387808 | ||
|
|
a6e74cdbe6 | ||
|
|
ff4da8a620 | ||
|
|
1558a2c65d | ||
|
|
66cad978c3 | ||
|
|
c11a0cef68 | ||
|
|
8cf2bf7ad8 | ||
|
|
ef5439e5ae | ||
|
|
c0b24b2f2e | ||
|
|
0d3a5f96f0 | ||
|
|
73dc3b3edd | ||
|
|
01bc89b82e | ||
|
|
2f663db52c | ||
|
|
e17fdf9e90 | ||
|
|
b695c5ae11 | ||
|
|
f5935b4006 | ||
|
|
eca6569bd3 | ||
|
|
51c241d1f8 | ||
|
|
83ead40b78 | ||
|
|
dea5c62a44 | ||
|
|
b3f902166b | ||
|
|
98964c5527 | ||
|
|
8ccc8dab49 | ||
|
|
e1d8881f57 | ||
|
|
9742a560f5 | ||
|
|
7a8b9dd341 | ||
|
|
063af711af | ||
|
|
aed71ed5e9 | ||
|
|
8b9237fd52 | ||
|
|
fd180a16d3 | ||
|
|
2a82f58195 | ||
|
|
34ffd2d6e2 | ||
|
|
62c9c130c4 | ||
|
|
fca23d88c3 | ||
|
|
3a179a3a40 | ||
|
|
fe17c447d9 | ||
|
|
b21a8dafa7 | ||
|
|
926a871dd4 | ||
|
|
79ef5031eb | ||
|
|
f0135eb332 | ||
|
|
566b25a35b | ||
|
|
e6748ea1ed | ||
|
|
5ab2c723ff | ||
|
|
cac8785aca | ||
|
|
0814780c2f | ||
|
|
753bce5e73 | ||
|
|
d04ad79eaf | ||
|
|
baca86bc3f | ||
|
|
adfff26d58 | ||
|
|
ef3a6963f3 | ||
|
|
19dba4a3ec | ||
|
|
49c1cda5c0 | ||
|
|
2ea749c7ea | ||
|
|
9d4b493ec0 | ||
|
|
7e3e4f81ac | ||
|
|
e740af05b0 | ||
|
|
4ed80fd071 | ||
|
|
59116a06b9 | ||
|
|
a409220e54 | ||
|
|
485f1c6acf | ||
|
|
6646a5cc72 | ||
|
|
2a76d3b153 | ||
|
|
7614e48322 | ||
|
|
631b8c3f58 | ||
|
|
b4b38a92ef | ||
|
|
c3769a6937 | ||
|
|
8d69852785 | ||
|
|
b83dd97762 | ||
|
|
22498dc871 | ||
|
|
96172840f1 | ||
|
|
e8880e4e98 | ||
|
|
26f6cda589 | ||
|
|
f0b45010ec | ||
|
|
ad3939a44f | ||
|
|
c9b23eb0e7 | ||
|
|
44f69f4dd7 | ||
|
|
c3fb6ccaaa | ||
|
|
b47a36a47b | ||
|
|
af9b438eec | ||
|
|
fa9bbf081b | ||
|
|
db3817a260 | ||
|
|
635befb4c8 | ||
|
|
a69daa2524 | ||
|
|
b563703701 | ||
|
|
0717f4689b | ||
|
|
bf30f6af4c | ||
|
|
dfcbdac347 | ||
|
|
8d12f3d5aa | ||
|
|
02f8ab5f87 | ||
|
|
b2b6b263bb | ||
|
|
e3f208f61f | ||
|
|
008d990f20 | ||
|
|
c038bb5794 | ||
|
|
efc5eaf9c7 | ||
|
|
c1e9d19809 | ||
|
|
85efb47a3b | ||
|
|
bc69693b6c | ||
|
|
c9c5a75b76 | ||
|
|
f50279a2e7 | ||
|
|
561c32b45d | ||
|
|
f42dc71972 | ||
|
|
ea67665bbf | ||
|
|
c3b4d4b7bb | ||
|
|
492dac719f | ||
|
|
9377bac5e7 | ||
|
|
e0bbfdda84 | ||
|
|
9618b5adef | ||
|
|
a3d609ebb1 | ||
|
|
2b0d94536f | ||
|
|
73ab3bc56d | ||
|
|
412c49a8ce | ||
|
|
6f29dfcbb8 | ||
|
|
f3ab82a73f | ||
|
|
6734c9ed3e | ||
|
|
71afe47675 | ||
|
|
7075377a63 | ||
|
|
7c1036df09 | ||
|
|
a312347589 | ||
|
|
f386c62dba | ||
|
|
30a81484d7 | ||
|
|
00a6b046a8 | ||
|
|
c156498c5c | ||
|
|
adea3e410f | ||
|
|
40940ae0b3 | ||
|
|
1bab28dad4 | ||
|
|
4838265589 | ||
|
|
f6d20a1a88 | ||
|
|
11be444d45 | ||
|
|
363bf85b5b | ||
|
|
720039ca6d | ||
|
|
4e9399e029 | ||
|
|
466ee3cbcf | ||
|
|
390d5d2fb4 | ||
|
|
b543df8299 | ||
|
|
de16c18961 | ||
|
|
2158309c51 | ||
|
|
430846d7f2 | ||
|
|
f4e362c477 | ||
|
|
7966bfb077 | ||
|
|
a4e04dd7b4 | ||
|
|
b161a74365 | ||
|
|
fd141ed6d7 | ||
|
|
e2ff0dd1cd | ||
|
|
6284c39eb3 | ||
|
|
7f0bdf8d63 | ||
|
|
015beff4cc | ||
|
|
92e43c25d4 | ||
|
|
69fe85274f | ||
|
|
0122ef9e83 | ||
|
|
c772c0e4e7 | ||
|
|
c171c06192 | ||
|
|
9365e6240b | ||
|
|
f8e3571c66 | ||
|
|
6ebcf65451 | ||
|
|
4844729e93 | ||
|
|
9686454161 | ||
|
|
167d79f0fa | ||
|
|
b1275edb63 | ||
|
|
e869aa644a | ||
|
|
68740b3c65 | ||
|
|
4caad9bf55 | ||
|
|
7629323548 | ||
|
|
681636cd68 | ||
|
|
2fdbff3d1c | ||
|
|
5c7df98768 | ||
|
|
efe38f2ee2 | ||
|
|
08031a2767 | ||
|
|
34a87cc94c | ||
|
|
d295d9f08e | ||
|
|
61d033f21e | ||
|
|
27315e33ab | ||
|
|
e8b5cd18bd | ||
|
|
1a77f41846 | ||
|
|
d1947f715c | ||
|
|
3f7971b7d2 | ||
|
|
2cbd5cb3e6 | ||
|
|
151b4d4c2d | ||
|
|
7a3fdefafb | ||
|
|
86c20d0519 | ||
|
|
464ec9d0bc | ||
|
|
35518a5fa0 | ||
|
|
d028c7942b | ||
|
|
585286a617 | ||
|
|
4904fd43e9 | ||
|
|
9e3f287c1f | ||
|
|
f5e4e26e08 | ||
|
|
6c5a39e164 | ||
|
|
7469fdb0d6 | ||
|
|
fcf9fd77d7 | ||
|
|
0589d9b872 | ||
|
|
6d4c80adff | ||
|
|
56dd470528 | ||
|
|
6741f53d87 | ||
|
|
19550c5417 | ||
|
|
cfa3dfaac7 | ||
|
|
0ed0bc1dd5 | ||
|
|
074743d6e8 | ||
|
|
8b446de059 | ||
|
|
f2e35f336f | ||
|
|
10a0fe2e71 | ||
|
|
c9add0d292 | ||
|
|
c239d09ea0 | ||
|
|
a652a5811b | ||
|
|
d2c92808f5 | ||
|
|
2464633431 | ||
|
|
15e76e88b4 | ||
|
|
fe1ac1bc1d | ||
|
|
9edd27b214 | ||
|
|
eb7e02ea1d | ||
|
|
53befc68c9 | ||
|
|
19f95d89ec | ||
|
|
ecb9b3b7f8 | ||
|
|
d619e36523 | ||
|
|
fa80d11960 | ||
|
|
2893d2b64f | ||
|
|
99b8df4e6f | ||
|
|
331655182d | ||
|
|
ec95330dcd | ||
|
|
3959863462 | ||
|
|
c00b2c9584 | ||
|
|
6c354f3987 | ||
|
|
3bd4d244a4 | ||
|
|
9888de25fe | ||
|
|
58c247b30c | ||
|
|
22bd10f3b1 | ||
|
|
f374b4775a | ||
|
|
ff213bbc5e | ||
|
|
56a4ca6e6a | ||
|
|
d6b38b6b1c | ||
|
|
04d06d386f | ||
|
|
aa26a780ed | ||
|
|
f73b93dbc2 | ||
|
|
c4a5ac892f | ||
|
|
f129ca0c61 | ||
|
|
941f0fbf8d | ||
|
|
6e37dca566 | ||
|
|
1cffa009fe | ||
|
|
83f4c9d101 | ||
|
|
c0c95da796 | ||
|
|
8b612a87c6 | ||
|
|
2ac95f446b | ||
|
|
c5eddd922d | ||
|
|
b58be2c073 | ||
|
|
a3d0b67777 | ||
|
|
0467d523c0 | ||
|
|
b0af054a95 | ||
|
|
6846f10510 | ||
|
|
e24f232f52 | ||
|
|
a1cc5d034e | ||
|
|
b478f54aea | ||
|
|
7bb380a086 | ||
|
|
228c351396 | ||
|
|
4bb675a530 | ||
|
|
8d62773570 | ||
|
|
9e26c57643 | ||
|
|
35bc502062 | ||
|
|
7189e1e280 | ||
|
|
4b4aa1cdbe | ||
|
|
3680282b30 | ||
|
|
a4d5fbae63 | ||
|
|
2943b87c83 | ||
|
|
9575506391 | ||
|
|
31c2449d6e | ||
|
|
70eafc4f6f | ||
|
|
38cbd2aeb2 | ||
|
|
ba762a9326 | ||
|
|
921ce59054 | ||
|
|
a7627ba39a | ||
|
|
34ef28dea5 | ||
|
|
0f2463dda2 | ||
|
|
5ec852637c | ||
|
|
ba8b0afe7a | ||
|
|
98d45a6a9b | ||
|
|
6bc808cb06 | ||
|
|
1325fef703 | ||
|
|
d2d0ef1803 | ||
|
|
d6fb60d512 | ||
|
|
ef35474f88 | ||
|
|
372891361c | ||
|
|
50c75d1f43 | ||
|
|
30f2a7b23b | ||
|
|
f2212a497b | ||
|
|
12e8cf008a | ||
|
|
9e8e87bbb2 | ||
|
|
76c4ebb36f | ||
|
|
9ff322eeed | ||
|
|
4afa49824e | ||
|
|
23402bf31b | ||
|
|
50be718b72 | ||
|
|
903e7db427 | ||
|
|
5e5e9e0803 | ||
|
|
ce27754b9d | ||
|
|
4254c0f5a9 | ||
|
|
7efc3ecaba | ||
|
|
2934b01d58 | ||
|
|
192e363701 | ||
|
|
1287365616 | ||
|
|
4fa539fbb2 | ||
|
|
28cdae4687 | ||
|
|
842e22915c | ||
|
|
bd150233ce | ||
|
|
24720b67da | ||
|
|
356d461123 | ||
|
|
3cbcc7b9f8 | ||
|
|
329f12a888 | ||
|
|
c0b2eec5de | ||
|
|
9b8ae25491 | ||
|
|
c0ee865e3e | ||
|
|
46ec2797ff | ||
|
|
fb2cdc8541 | ||
|
|
850ef70496 | ||
|
|
9d3c388664 | ||
|
|
f2b679f602 | ||
|
|
b9d97dffe7 | ||
|
|
7ff0466c2c | ||
|
|
a135325185 | ||
|
|
a6c8f0d300 | ||
|
|
046750354e | ||
|
|
e23d703873 | ||
|
|
ca2d2520d2 | ||
|
|
bc16f902d1 | ||
|
|
f5ae888597 | ||
|
|
376e3af058 | ||
|
|
af63c0a9e1 | ||
|
|
7d149ebec4 | ||
|
|
37c809471b | ||
|
|
12baceb859 | ||
|
|
3e6b3c8c40 | ||
|
|
25f2021711 | ||
|
|
8b30f7dbb5 | ||
|
|
76b35dfeb0 | ||
|
|
4518d5831b | ||
|
|
066250851f | ||
|
|
341a5196ba | ||
|
|
c878a89e95 | ||
|
|
c72d59a3f1 | ||
|
|
995e2657bb | ||
|
|
fe6d6397d6 | ||
|
|
c7f52bcee4 | ||
|
|
fb4d8a6d14 | ||
|
|
929f9a7ad2 | ||
|
|
48ce5bf6e6 | ||
|
|
617a518714 | ||
|
|
480f45f2f3 | ||
|
|
0e8b01c1b5 | ||
|
|
b750d6772f | ||
|
|
7ce172a497 | ||
|
|
821dfe5b98 | ||
|
|
6c57b2f8f9 | ||
|
|
6e0c9d159d | ||
|
|
a90a7dbec7 | ||
|
|
2250bd58a2 | ||
|
|
126be45ed1 | ||
|
|
e021a55abd | ||
|
|
2ad6254894 | ||
|
|
5bd97c2f63 | ||
|
|
570d1f2271 | ||
|
|
613e9ef701 | ||
|
|
5080c6ffc5 | ||
|
|
f16bc12b7d | ||
|
|
d2f096187c | ||
|
|
bd1e61befd | ||
|
|
b97165bec9 | ||
|
|
8b0f07b2dc | ||
|
|
c0ba45f6de | ||
|
|
017c898ed0 | ||
|
|
b193a0c9fd | ||
|
|
db0c7b0562 | ||
|
|
85b8e91b57 | ||
|
|
87301ca154 | ||
|
|
650f5b784d | ||
|
|
c8dd9eefa6 | ||
|
|
0bdf15977f | ||
|
|
8a844f9cca | ||
|
|
8fbde84380 | ||
|
|
00826d3327 | ||
|
|
d5572e322f | ||
|
|
37814111de | ||
|
|
1417888a89 | ||
|
|
5d846c3ca9 | ||
|
|
5dd0477440 | ||
|
|
c6f823f855 | ||
|
|
2af7c24e79 | ||
|
|
94ccfefa84 | ||
|
|
46f3bec37e | ||
|
|
f5d2e0db29 | ||
|
|
88ee56f471 | ||
|
|
8436154276 | ||
|
|
150b25f29e | ||
|
|
f4c50105ba | ||
|
|
3395bedc2b | ||
|
|
7a14210e2b | ||
|
|
d206b67ca8 | ||
|
|
0bfef9008b | ||
|
|
401e542fcd | ||
|
|
210514f74b | ||
|
|
dd838d4ad3 | ||
|
|
5f2d19c7aa | ||
|
|
d5ae87b5ca | ||
|
|
95e7c866cf | ||
|
|
02028eb1ad | ||
|
|
2effdb04bd | ||
|
|
ab31e3e3bb | ||
|
|
37b1432514 | ||
|
|
6e8bc337aa | ||
|
|
8347566815 | ||
|
|
ede09a03db | ||
|
|
849d60253c | ||
|
|
b5660c8a92 | ||
|
|
54263bb5a7 | ||
|
|
4069a9f7d5 | ||
|
|
77467eaf0c | ||
|
|
caa030714b | ||
|
|
facbc78e0a | ||
|
|
6488dcbb01 | ||
|
|
37265b109a | ||
|
|
ebe7342d10 | ||
|
|
523bbef034 | ||
|
|
84e127a637 | ||
|
|
4a091df8cc | ||
|
|
92a171ce53 | ||
|
|
1b1e46ff6c | ||
|
|
3370d9af15 | ||
|
|
9306de79ad | ||
|
|
ff7a54add8 | ||
|
|
c3d1157696 | ||
|
|
d0f03cb148 | ||
|
|
9b7c9f0fb6 | ||
|
|
e508b6df0d | ||
|
|
710b85be70 | ||
|
|
66455b708a | ||
|
|
8b8000b98a | ||
|
|
54236df6e0 | ||
|
|
6cd2a48910 | ||
|
|
fe08b96844 | ||
|
|
4d78901420 | ||
|
|
c06468025c | ||
|
|
148e539025 | ||
|
|
92b96ff30d | ||
|
|
778df0c93b | ||
|
|
9d18ecc5cb | ||
|
|
5f2455c502 | ||
|
|
6bc67609a3 | ||
|
|
1bd3945dd9 | ||
|
|
168f4b1e6b | ||
|
|
8a8827c980 | ||
|
|
21a968b84d | ||
|
|
84fab84b3f | ||
|
|
3d65c030a8 | ||
|
|
59097bab20 | ||
|
|
4cbacd9261 | ||
|
|
655102fc7d | ||
|
|
f718f46545 | ||
|
|
71d4e2c320 | ||
|
|
41241d7500 | ||
|
|
b90c9836cb | ||
|
|
aff3fcf76d | ||
|
|
ec2aa4063a | ||
|
|
718f2e01f7 | ||
|
|
ba9f7fb1b5 | ||
|
|
0ac6b3e8f3 | ||
|
|
dddad1c2ca | ||
|
|
95bfff20a4 | ||
|
|
7a6f0def85 | ||
|
|
103d4d76be | ||
|
|
db9414a756 | ||
|
|
5a0bf1bb5f | ||
|
|
444c37fe2e | ||
|
|
d3d735370f | ||
|
|
852e93aa74 | ||
|
|
dc1be2efe8 | ||
|
|
f2734ac608 | ||
|
|
57ca49dc5c | ||
|
|
5ef3134a4d | ||
|
|
5b91642883 | ||
|
|
1eb5abb8db | ||
|
|
a65e1bf7e5 | ||
|
|
3f98202b0f | ||
|
|
e7727c39f3 | ||
|
|
17e5637664 | ||
|
|
7fd9b897c2 | ||
|
|
f8967aa207 | ||
|
|
5145324806 | ||
|
|
4233fb6270 | ||
|
|
d11b19fd2b | ||
|
|
684c568033 | ||
|
|
ab4fb7b3ad | ||
|
|
91017dbedb | ||
|
|
f5e8a051a8 | ||
|
|
1d695f6db4 | ||
|
|
d4bdfd0592 |
No files matched your search
@@ -43,7 +43,7 @@ jobs:
|
||||
distrobox upgrade steamrt4
|
||||
distrobox enter --name steamrt4 -- sudo apt-get install -y \
|
||||
git cmake ninja-build ccache \
|
||||
lld clang \
|
||||
lld clang clang-tools \
|
||||
libclang-dev llvm-dev \
|
||||
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
|
||||
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
|
||||
|
||||
@@ -24,7 +24,7 @@ runs:
|
||||
cmake -S . -B build_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=Data/CMake/toolchain_mingw.cmake \
|
||||
-DMINGW_TRIPLE=${_cc}-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja \
|
||||
-DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
|
||||
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none
|
||||
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none -DRANGES_NATIVE=OFF
|
||||
|
||||
- name: Build
|
||||
shell: bash
|
||||
|
||||
+3
-3
@@ -31,12 +31,12 @@ build:
|
||||
- apt-get -y update
|
||||
- apt-get install -y
|
||||
git cmake ninja-build ccache
|
||||
lld clang
|
||||
lld clang clang-tools
|
||||
libclang-dev llvm-dev
|
||||
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross
|
||||
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
|
||||
- cmake -E make_directory build/
|
||||
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none . -B build/
|
||||
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none -DRANGES_NATIVE=OFF . -B build/
|
||||
- cmake --build build/ --config Release
|
||||
- DESTDIR=$(pwd)/install/ cmake --build build/ --config Release -t install
|
||||
|
||||
@@ -57,7 +57,7 @@ promote:
|
||||
- arm64
|
||||
- aarch64
|
||||
rules:
|
||||
- if: '$PROMOTE_BRANCH'
|
||||
- if: $PROMOTE_BRANCH && $CI_COMMIT_BRANCH == 'main'
|
||||
before_script:
|
||||
- apt-get -y update
|
||||
- apt-get install -y tmux curl
|
||||
|
||||
+43
-13
@@ -195,6 +195,9 @@ if (ENABLE_GDB_SYMBOLS)
|
||||
endif()
|
||||
|
||||
add_compile_definitions(_LARGEFILE64_SOURCE)
|
||||
if (WIN32)
|
||||
add_compile_definitions(UNICODE _UNICODE)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
@@ -364,7 +367,10 @@ include(LinkerGC)
|
||||
|
||||
## Externals ##
|
||||
|
||||
find_package(unordered_dense QUIET CONFIG)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(unordered_dense QUIET CONFIG)
|
||||
endif()
|
||||
|
||||
if (NOT unordered_dense_FOUND)
|
||||
add_subdirectory(External/unordered_dense)
|
||||
endif()
|
||||
@@ -375,8 +381,10 @@ if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ZYDIS)
|
||||
find_package(Zycore 1.5 MODULE QUIET)
|
||||
find_package(Zydis 4.0 MODULE QUIET)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(Zycore 1.5 MODULE QUIET)
|
||||
find_package(Zydis 4.0 MODULE QUIET)
|
||||
endif()
|
||||
|
||||
if (TARGET Zydis::Zydis AND TARGET Zycore::Zycore)
|
||||
message(STATUS "Using system Zydis")
|
||||
@@ -397,7 +405,7 @@ find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
|
||||
if (NOT CMAKE_CROSSCOMPILING)
|
||||
if (NOT CMAKE_CROSSCOMPILING AND NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(xxhash MODULE QUIET)
|
||||
endif()
|
||||
|
||||
@@ -425,14 +433,22 @@ else ()
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
if (MINGW OR BUILD_STEAM_SUPPORT)
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
else()
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(range-v3 QUIET)
|
||||
endif()
|
||||
|
||||
find_package(range-v3 QUIET)
|
||||
if (NOT range-v3_FOUND)
|
||||
add_subdirectory(External/range-v3/)
|
||||
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
|
||||
@@ -477,12 +493,20 @@ endif()
|
||||
|
||||
set(FEX_TUNE_COMPILE_FLAGS)
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
set(TUNE_ARCH_STRING "${TUNE_ARCH}")
|
||||
if(ARCHITECTURE_arm64)
|
||||
set(TUNE_ARCH_STRING "${TUNE_ARCH}+crc")
|
||||
endif()
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH_STRING}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH_STRING}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH_STRING}' but the compiler doesn't support this")
|
||||
endif()
|
||||
elseif(ARCHITECTURE_arm64)
|
||||
# Need to always append crc
|
||||
check_cxx_compiler_flag("-march=armv8-a+crc" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=armv8-a+crc")
|
||||
endif()
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
@@ -593,6 +617,12 @@ if (BUILD_TESTING)
|
||||
execute_process(COMMAND "nproc" OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE TEST_JOB_COUNT)
|
||||
endif()
|
||||
set(TEST_JOB_FLAG "-j${TEST_JOB_COUNT}")
|
||||
|
||||
# Runs the whole test suite, with large pages enabled to reduce the cost of fork() in ctest.
|
||||
add_custom_target(tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
USES_TERMINAL
|
||||
COMMAND ${CMAKE_COMMAND} -E env GLIBC_TUNABLES=glibc.malloc.hugetlb=1 ctest "--progress" "--timeout" "302" ${TEST_JOB_FLAG})
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/SoftFloat-3e/)
|
||||
|
||||
@@ -270,6 +270,9 @@ public:
|
||||
void fcvtxnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFloatConvertOdd(0b00, 0b10, pg, zn, zd);
|
||||
}
|
||||
void bfcvtnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFloatConvertOdd(0b10, 0b10, pg, zn, zd);
|
||||
}
|
||||
///< Size is destination size
|
||||
void fcvtnt(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i32Bit || size == SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
@@ -292,8 +295,6 @@ public:
|
||||
SVEFloatConvertOdd(ConvertedSrcSize, ConvertedDestSize, pg, zn, zd);
|
||||
}
|
||||
|
||||
// XXX: BFCVTNT
|
||||
|
||||
// SVE2 floating-point pairwise operations
|
||||
void faddp(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn, ZRegister zm) {
|
||||
SVEFloatPairwiseArithmetic(0b000, size, pg, zd, zn, zm);
|
||||
@@ -2312,15 +2313,15 @@ public:
|
||||
|
||||
// SVE floating-point convert precision
|
||||
void fcvt(SubRegSize to, SubRegSize from, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && from != SubRegSize::i8Bit, "Can't use 8-bit element size");
|
||||
SVEFPConvertPrecision(to, from, zd, pg, zn);
|
||||
}
|
||||
void fcvtx(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
uint32_t Instr = 0b0110'0101'0000'1010'1010'0000'0000'0000;
|
||||
Instr |= pg.Idx() << 10;
|
||||
Instr |= zn.Idx() << 5;
|
||||
Instr |= zd.Idx();
|
||||
dc32(Instr);
|
||||
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i8Bit, zd, pg, zn);
|
||||
}
|
||||
void bfcvt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i32Bit, zd, pg, zn);
|
||||
}
|
||||
|
||||
// SVE floating-point unary operations
|
||||
@@ -3847,14 +3848,19 @@ private:
|
||||
|
||||
void SVEFPConvertPrecision(SubRegSize to, SubRegSize from, ZRegister zd, PRegister pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && to != SubRegSize::i128Bit && from != SubRegSize::i8Bit && from != SubRegSize::i128Bit,
|
||||
"Can't use 8-bit or 128-bit element size");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i128Bit && from != SubRegSize::i128Bit, "Can't use 128-bit element size");
|
||||
|
||||
// Encodings for the to and from sizes can get a little funky
|
||||
// depending on what is being converted to/from.
|
||||
const uint32_t op = [&] {
|
||||
switch (from) {
|
||||
case SubRegSize::i8Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i32Bit: return 0x00020000U;
|
||||
default: return UINT32_MAX;
|
||||
}
|
||||
}
|
||||
|
||||
case SubRegSize::i16Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i32Bit: return 0x00810000U;
|
||||
@@ -3866,6 +3872,7 @@ private:
|
||||
case SubRegSize::i32Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i16Bit: return 0x00800000U;
|
||||
case SubRegSize::i32Bit: return 0x00820000U;
|
||||
case SubRegSize::i64Bit: return 0x00C30000U;
|
||||
default: return UINT32_MAX;
|
||||
}
|
||||
|
||||
@@ -9,8 +9,8 @@ set(CMAKE_AR ${MINGW_TRIPLE}-ar)
|
||||
# Compile everything as static to avoid requiring the MinGW runtime libraries, force page aligned sections so that
|
||||
# debug symbols work correctly, and disable loop alignment to workaround an LLVM bug
|
||||
# (https://github.com/llvm/llvm-project/issues/47432)
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_C_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
set(CMAKE_CXX_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
set(CMAKE_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 450bd22322...ee2ec5fd83.
+50
-53
@@ -210,56 +210,53 @@ click==8.1.7 \
|
||||
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
|
||||
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
|
||||
# via black
|
||||
cryptography==48.0.0 \
|
||||
--hash=sha256:0890f502ddf7d9c6426129c3f49f5c0a39278ed7cd6322c8755ffca6ee675a13 \
|
||||
--hash=sha256:0c558d2cdffd8f4bbb30fc7134c74d2ca9a476f830bb053074498fbc86f41ed6 \
|
||||
--hash=sha256:16cd65b9330583e4619939b3a3843eec1e6e789744bb01e7c7e2e62e33c239c8 \
|
||||
--hash=sha256:18349bbc56f4743c8b12dc32e2bccb2cf83ee8b69a3bba74ef8ae857e26b3d25 \
|
||||
--hash=sha256:1e2d54c8be6152856a36f0882ab231e70f8ec7f14e93cf87db8a2ed056bf160c \
|
||||
--hash=sha256:22a5cb272895dce158b2cacdfdc3debd299019659f42947dbdac6f32d68fe832 \
|
||||
--hash=sha256:27241b1dc9962e056062a8eef1991d02c3a24569c95975bd2322a8a52c6e5e12 \
|
||||
--hash=sha256:2b4d59804e8408e2fea7d1fbaf218e5ec984325221db76e6a241a9abd6cdd95c \
|
||||
--hash=sha256:2eb992bbd4661238c5a397594c83f5b4dc2bc5b848c365c8f991b6780efcc5c7 \
|
||||
--hash=sha256:369a6348999f94bbd53435c894377b20ab95f25a9065c283570e70150d8abc3c \
|
||||
--hash=sha256:3cb07a3ed6431663cd321ea8a000a1314c74211f823e4177fefa2255e057d1ec \
|
||||
--hash=sha256:40ba1f85eaa6959837b1d51c9767e230e14612eea4ef110ee8854ada22da1bf5 \
|
||||
--hash=sha256:4defde8685ae324a9eb9d818717e93b4638ef67070ac9bc15b8ca85f63048355 \
|
||||
--hash=sha256:55b7718303bf06a5753dcdccf2f3945cf18ad7bffde41b61226e4db31ab89a9c \
|
||||
--hash=sha256:561215ea3879cb1cbbf272867e2efda62476f240fb58c64de6b393ae19246741 \
|
||||
--hash=sha256:58d00498e8933e4a194f3076aee1b4a97dfec1a6da444535755822fe5d8b0b86 \
|
||||
--hash=sha256:59baa2cb386c4f0b9905bd6eb4c2a79a69a128408fd31d32ca4d7102d4156321 \
|
||||
--hash=sha256:5a5ed8fde7a1d09376ca0b40e68cd59c69fe23b1f9768bd5824f54681626032a \
|
||||
--hash=sha256:5b012212e08b8dd5edc78ef54da83dd9892fd9105323b3993eff6bea65dc21d7 \
|
||||
--hash=sha256:5c3932f4436d1cccb036cb0eaef46e6e2db91035166f1ad6505c3c9d5a635920 \
|
||||
--hash=sha256:614d0949f4790582d2cc25553abd09dd723025f0c0e7c67376a1d77196743d6e \
|
||||
--hash=sha256:76341972e1eff8b4bea859f09c0d3e64b96ce931b084f9b9b7db8ef364c30eff \
|
||||
--hash=sha256:77a2ccbbe917f6710e05ba9adaa25fb5075620bf3ea6fb751997875aff4ae4bd \
|
||||
--hash=sha256:7995ef305d7165c3f11ae07f2517e5a4f1d5c18da1376a0a9ed496336b69e5f3 \
|
||||
--hash=sha256:7ce4bfae76319a532a2dc68f82cc32f5676ee792a983187dac07183690e5c66f \
|
||||
--hash=sha256:7e8eac43dfca5c4cccc6dad9a80504436fca53bb9bc3100a2386d730fbe6b602 \
|
||||
--hash=sha256:84cf79f0dc8b36ac5da873481716e87aef31fcfa0444f9e1d8b4b2cece142855 \
|
||||
--hash=sha256:8c7378637d7d88016fa6791c159f698b3d3eed28ebf844ac36b9dc04a14dae18 \
|
||||
--hash=sha256:8cd666227ef7af430aa5914a9910e0ddd703e75f039cef0825cd0da71b6b711a \
|
||||
--hash=sha256:906cbf0670286c6e0044156bc7d4af9cbb0ef6db9f73e52c3ec56ba6bdde5336 \
|
||||
--hash=sha256:9071196d81abc88b3516ac8cdfad32e2b66dd4a5393a8e68a961e9161ddc6239 \
|
||||
--hash=sha256:9249e3cd978541d665967ac2cb2787fd6a62bddf1e75b3e347a594d7dacf4f74 \
|
||||
--hash=sha256:984a20b0f62a26f48a3396c72e4bc34c66e356d356bf370053066b3b6d54634a \
|
||||
--hash=sha256:9be5aafa5736574f8f15f262adc81b2a9869e2cfe9014d52a44633905b40d52c \
|
||||
--hash=sha256:9c459db21422be75e2809370b829a87eb37f74cd785fc4aa9ea1e5f43b47cda4 \
|
||||
--hash=sha256:9ccdac7d40688ecb5a3b4a604b8a88c8002e3442d6c60aead1db2a89a041560c \
|
||||
--hash=sha256:a0e692c683f4df67815a2d258b324e66f4738bd7a96a218c826dce4f4bd05d8f \
|
||||
--hash=sha256:a5da777e32ffed6f85a7b2b3f7c5cbc88c146bfcd0a1d7baf5fcc6c52ee35dd4 \
|
||||
--hash=sha256:a64697c641c7b1b2178e573cbc31c7c6684cd56883a478d75143dbb7118036db \
|
||||
--hash=sha256:ad64688338ed4bc1a6618076ba75fd7194a5f1797ac60b47afe926285adb3166 \
|
||||
--hash=sha256:bd72e68b06bb1e96913f97dd4901119bc17f39d4586a5adf2d3e47bc2b9d58b5 \
|
||||
--hash=sha256:c17dfe85494deaeddc5ce251aebd1d60bbe6afc8b62071bb0b469431a000124f \
|
||||
--hash=sha256:c18684a7f0cc9a3cb60328f496b8e3372def7c5d2df39ac267878b05565aaaae \
|
||||
--hash=sha256:cc90c0b39b2e3c65ef52c804b72e3c58f8a04ab2a1871272798e5f9572c17d20 \
|
||||
--hash=sha256:db63bf618e5dea46c07de12e900fe1cdd2541e6dc9dbae772a70b7d4d4765f6a \
|
||||
--hash=sha256:ea8990436d914540a40ab24b6a77c0969695ed52f4a4874c5137ccf7045a7057 \
|
||||
--hash=sha256:ecde28a596bead48b0cfd2a1b4416c3d43074c2d785e3a398d7ec1fc4d0f7fbb \
|
||||
--hash=sha256:f5333311663ea94f75dd408665686aaf426563556bb5283554a3539177e03b8c \
|
||||
--hash=sha256:fdfef35d751d510fcef5252703621574364fec16418c4a1e5e1055248401054b
|
||||
cryptography==50.0.0 \
|
||||
--hash=sha256:031e2d5dd4bb9caa3ca9c82e5a197fd8ae680232cee62603d1a813f3f07e3d03 \
|
||||
--hash=sha256:06a32a980526a6ab9a4b9bf8f7385800791e2bb960903cb6b530e4817509a3b7 \
|
||||
--hash=sha256:07479a1cb08219ab719147e742e76090c9c773321959bb94946fffdd397a6437 \
|
||||
--hash=sha256:07949c449a1abcf60d1ee6e88956d89404c7df3c8258f46589e912988e551987 \
|
||||
--hash=sha256:105110f43a471dbd0060b9c9516cb8a6a79233631a04cc2ba16f28323ac6e025 \
|
||||
--hash=sha256:11b74db56cdbe3cdee6e3f6982ecb70334fa10dce99ed58bf7894aaaa3b2a037 \
|
||||
--hash=sha256:12b9c6996425c76ea6c457ace4f3073e715b8c545add07cd1a8f3a4f90691269 \
|
||||
--hash=sha256:1489e263a8048bb8b6a8bac662eb2d402ea5d2b7b4699b72f385f1e2772db105 \
|
||||
--hash=sha256:19736989797678c6af1e55cd49055cdbcb55d8f6b5583ac5335f933aba9101dc \
|
||||
--hash=sha256:1b4a266766514614f8aa60416e71f2fc6e575d36e7bdc90f644fadb2f4b75b95 \
|
||||
--hash=sha256:2a8183b489dc1f7f80f135780fadc1108f14b31b8a40411c7a5b17425f65f28b \
|
||||
--hash=sha256:37fdb0d0111f1e2ff07139dfb79f1b49531f8e213c46f1163dd7642979b58c47 \
|
||||
--hash=sha256:3f5735ffe4996d28b809371756219f5354864902a3b9e7c0b9ee87041209fc9c \
|
||||
--hash=sha256:49e7d93abdbd2990caced757e5fade25302f719c3c8fb6e6fff2dde98999fc41 \
|
||||
--hash=sha256:5e34edd123674534acd70147f0ca331eaa2c74e6325fb2028c886aa26ba0b68c \
|
||||
--hash=sha256:62598a8a57f815db4c6259a4e97d857dab56697e7de8e8ab02352ab74da1995d \
|
||||
--hash=sha256:65c2c3add92b45fd0709db8594536aea39c2a67af0e27ffcf049c498501140b7 \
|
||||
--hash=sha256:6ba6a53445bd3cfa809ef3ef5f1589aa6ba08784a1d962bf47d0940e871dab1c \
|
||||
--hash=sha256:6e7d61120573a7f2cd94cc095f9e81f6967c61ccdf194285aa143ecec8e0b708 \
|
||||
--hash=sha256:7cec5b856506da6defb290f30c9ee687d5f5e8cb0bd3f6459dde43b0b4fa40ef \
|
||||
--hash=sha256:80b63928fa35083b33966ce1efb70e5b9607181e49dcd1c22c8c005e319f667f \
|
||||
--hash=sha256:82148ec5bddac30b51a5b3c1945075f896fa022cb93f8e4a01e9f6ee95292c5f \
|
||||
--hash=sha256:828743d939e9629bc267b8e2d08d8bb67cd4319c771a33d4b18b22dd8fb7440a \
|
||||
--hash=sha256:8d89f3976b10b4ce31118de72329025f70d2c6ead14a8217c5514dd2c6d5a78f \
|
||||
--hash=sha256:8eb5e1172eb569ea8a872796576e6a67c276351728b6455d5beb01242b027c6a \
|
||||
--hash=sha256:900131fafd8aead39ac7dd3a7e833be754c17a95cfd91221636949fe4eb0aa8a \
|
||||
--hash=sha256:910d11e1a385c654bf738bf3e6b8e6ed5de0f5610fcae2be9e5b398d8081d20e \
|
||||
--hash=sha256:910e1d2668e7de9648f2bcee30e180db2a6b15c30f887d7c4c93ddf96e3992e3 \
|
||||
--hash=sha256:9aa87839c383bdbab6ef865787a1fb877af8dd03464c4400322726feaaadfc6d \
|
||||
--hash=sha256:a1b30560f2acc95aa8b2e06e716a13dbfc97314747b80d9707e307f77b40d6b3 \
|
||||
--hash=sha256:a91296cb61e8df6f86d0c19cc4068228da256bf59bf86049fbd821084565327f \
|
||||
--hash=sha256:b42a28c1844fd9de8f3f7d540e36b66f3a9c83fceac7170ebc7a6a19edd9dcae \
|
||||
--hash=sha256:bd1c592e4d5974f0d08d4888e432157adba757c66da0246918e43677fafa2d30 \
|
||||
--hash=sha256:c87f62a3d3b9888ed0fdde100ec06aa61ca9cd44bad9057d1dff9a516b5f5bb9 \
|
||||
--hash=sha256:c99c003e088647b8a5b7c145d6f78c335f6348332b62e142d411c4b63d1460b9 \
|
||||
--hash=sha256:ccdc4a71a4dabae05de219404f9f4abc38e3b58422177ff93d0da05967dafa07 \
|
||||
--hash=sha256:d24fead1d4d076e1bfb006dcec392074a3cd8d7b4fc8a595aa64073b2b7a96ba \
|
||||
--hash=sha256:d58c3db7cd6eed54e6c06744db55456b65ebd7492ddeae9c1e93cfca7aa857d3 \
|
||||
--hash=sha256:d764dcf130c428ef66786f866dd750f53182bc608813489915e9fc106bb0c82f \
|
||||
--hash=sha256:df2a58a472f332225671c35b0a830208b86d004f82baa8530fa3782c85646533 \
|
||||
--hash=sha256:e722f16708d854fe924790e051061f6704a472c3bac347b6fd88033ea8dd0dc5 \
|
||||
--hash=sha256:ecfed7367f965a0328cfbdd70da860f15441f002f613185668c6e6ebf5a0ac11 \
|
||||
--hash=sha256:eeac2acb5a20ed25e0ad6d1df9891a520b78b404266b6d11778f25d5d691a6c9 \
|
||||
--hash=sha256:f59e38625469987d7ef6d495323c55e7db6c212eaf6112267e0d3b565a2e9c9f \
|
||||
--hash=sha256:f89831ef99dd7dd169ab06d63a831adb9e20a87aac6d380266bbda5823349169 \
|
||||
--hash=sha256:fd9192b7b70c573d7f214eb1ae35e00d359f6f5e4b27c7e21e30de1fc6204645
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pyjwt
|
||||
@@ -311,9 +308,9 @@ pygithub==2.6.1 \
|
||||
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
|
||||
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
|
||||
# via -r requirements_formatting.txt.in
|
||||
pyjwt==2.12.1 \
|
||||
--hash=sha256:28ca37c070cad8ba8cd9790cd940535d40274d22f80ab87f3ac6a713e6e8454c \
|
||||
--hash=sha256:c74a7a2adf861c04d002db713dd85f84beb242228e671280bf709d765b03672b
|
||||
pyjwt==2.15.1 \
|
||||
--hash=sha256:42d59d631f7768a1028a64c7ff581a9bf7519804daf91fc5b6c56e30eec5e193 \
|
||||
--hash=sha256:4f259e80cdfb6b3fc18a7de51fd1ef9ec79652f25019bae68975ca2468a34df8
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
black>=26.3.1
|
||||
darker==2.1.1
|
||||
PyGithub==2.6.1
|
||||
cryptography>=46.0.7
|
||||
cryptography>=50.0.0
|
||||
urllib3>=2.7.0
|
||||
requests>=2.33.0
|
||||
idna>=3.15
|
||||
certifi>=2024.7.4
|
||||
PyNaCl>=1.6.2
|
||||
PyJWT>=2.12.1
|
||||
PyJWT>=2.15.1
|
||||
Vendored
+1
-1
Submodule External/fmt updated: 407c905e45...c07e2aa4b1.
Vendored
+1
-1
Submodule External/rpmalloc updated: 1d85c246cd...09142d7264.
Vendored
+1
-1
Submodule External/vixl updated: 5f418449c4...585d860b12.
+2
-2
@@ -8,8 +8,8 @@ This project aims to provide a fast and functional x86-64 emulation library that
|
||||
* Support a tiered recompiler to allow for fast runtime performance
|
||||
* Support offline compilation and offline tooling for inspection and performance analysis
|
||||
* Support threaded emulation. Including emulating x86-64's strong memory model on weak memory model architectures
|
||||
* Support a significant portion of the x86-64 instruction space.
|
||||
* Including MMX, SSE, SSE2, SSE3, SSSE3, and SSE4*
|
||||
* Support a majority of the x86-64 instruction space.
|
||||
* Including MMX, SSE, SSE2, SSE3, SSSE3, SSE4*, AVX, AVX2, F16C, and AVX-VNNI
|
||||
* Support fallback routines for uncommonly used x86-64 instructions
|
||||
* Including x87 and 3DNow!
|
||||
* Only support userspace emulation.
|
||||
|
||||
@@ -407,6 +407,32 @@ def print_parse_enum_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_affects_codegen_options(options, unnamed_options):
|
||||
output_argloader.write("#ifdef CONFIG_AFFECTSCODEGEN\n")
|
||||
output_argloader.write("#undef CONFIG_AFFECTSCODEGEN\n")
|
||||
|
||||
TotalConfigOptions = 0
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
TotalConfigOptions += 1
|
||||
for op_group, group_vals in unnamed_options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
TotalConfigOptions += 1
|
||||
|
||||
output_argloader.write("constexpr static std::array<bool, {}> Config_AffectsCodeGen = {{{{\n".format(TotalConfigOptions))
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
|
||||
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
|
||||
|
||||
for op_group, group_vals in unnamed_options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
|
||||
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
|
||||
output_argloader.write("}};\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
if (len(sys.argv) < 5):
|
||||
sys.exit()
|
||||
|
||||
@@ -451,4 +477,6 @@ print_parse_jsonloader_options(options);
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
print_affects_codegen_options(options, unnamed_options);
|
||||
|
||||
output_argloader.close()
|
||||
@@ -251,6 +251,10 @@ def parse_ops(ops):
|
||||
|
||||
if "Desc" in op_val:
|
||||
OpDef.Desc = op_val["Desc"]
|
||||
if not isinstance(OpDef.Desc, list):
|
||||
ExitError(f"Desc field for op {OpDef.Name} must be an array of strings")
|
||||
if not all(isinstance(item, str) for item in OpDef.Desc):
|
||||
ExitError(f"Desc field for op {OpDef.Name} must only contain strings")
|
||||
|
||||
if "DynamicDispatch" in op_val:
|
||||
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
|
||||
@@ -603,77 +607,77 @@ def print_validation(op):
|
||||
def print_ir_allocator_helpers():
|
||||
output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n")
|
||||
|
||||
output_file.write("\ttemplate <class T>\n")
|
||||
output_file.write("\tstruct Wrapper final {\n")
|
||||
output_file.write("\t\tT *first;\n")
|
||||
output_file.write("\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n")
|
||||
output_file.write("\n")
|
||||
output_file.write("\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n")
|
||||
output_file.write("\t\toperator OrderedNode *() { return Node; }\n")
|
||||
output_file.write("\t\toperator const OrderedNode *() const { return Node; }\n")
|
||||
output_file.write("\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n")
|
||||
output_file.write("\t};\n")
|
||||
output_file.write("\ttemplate <class T>\n"
|
||||
"\tstruct Wrapper final {\n"
|
||||
"\t\tT *first;\n"
|
||||
"\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n"
|
||||
"\n"
|
||||
"\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n"
|
||||
"\t\toperator OrderedNode *() { return Node; }\n"
|
||||
"\t\toperator const OrderedNode *() const { return Node; }\n"
|
||||
"\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n"
|
||||
"\t};\n")
|
||||
|
||||
output_file.write("\ttemplate <class T>\n")
|
||||
output_file.write("\tusing IRPair = Wrapper<T>;\n\n")
|
||||
output_file.write("\ttemplate <class T>\n"
|
||||
"\tusing IRPair = Wrapper<T>;\n\n")
|
||||
|
||||
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n")
|
||||
output_file.write("\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n")
|
||||
output_file.write("\t\tmemset(Op, 0, HeaderSize);\n")
|
||||
output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n")
|
||||
output_file.write("\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n"
|
||||
"\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n"
|
||||
"\t\tmemset(Op, 0, HeaderSize);\n"
|
||||
"\t\tOp->Op = IROps::OP_DUMMY;\n"
|
||||
"\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n")
|
||||
output_file.write("\tT *AllocateOrphanOp() {\n")
|
||||
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
|
||||
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
|
||||
output_file.write("\t\tmemset(Op, 0, Size);\n")
|
||||
output_file.write("\t\tOp->Header.Op = T2;\n")
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n"
|
||||
"\tT *AllocateOrphanOp() {\n"
|
||||
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
|
||||
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
|
||||
"\t\tmemset(Op, 0, Size);\n"
|
||||
"\t\tOp->Header.Op = T2;\n"
|
||||
"\t\treturn Op;\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n")
|
||||
output_file.write("\tIRPair<T> AllocateOp() {\n")
|
||||
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
|
||||
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
|
||||
output_file.write("\t\tmemset(Op, 0, Size);\n")
|
||||
output_file.write("\t\tOp->Header.Op = T2;\n")
|
||||
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n"
|
||||
"\tIRPair<T> AllocateOp() {\n"
|
||||
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
|
||||
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
|
||||
"\t\tmemset(Op, 0, Size);\n"
|
||||
"\t\tOp->Header.Op = T2;\n"
|
||||
"\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn HeaderOp->Size;\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn HeaderOp->ElementSize;\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n")
|
||||
output_file.write("\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n"
|
||||
"\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n"
|
||||
"\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn GetHasDest(HeaderOp->Op);\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn GetHasDest(HeaderOp->Op);\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn HeaderOp->Op;\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n"
|
||||
"\t\treturn GetRegClass(GetOpType(Op));\n"
|
||||
"\t}\n\n")
|
||||
|
||||
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn IR::GetName(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n"
|
||||
"\t\treturn IR::GetName(GetOpType(Op));\n"
|
||||
"\t}\n\n")
|
||||
|
||||
# Generate helpers with operands
|
||||
for op in IROps:
|
||||
|
||||
@@ -4,6 +4,7 @@ set(FEXCORE_BASE_SRCS
|
||||
Interface/Config/Config.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/FileUtils.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/SpinWaitLock.cpp
|
||||
@@ -18,12 +19,14 @@ set(SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/DiskCache.cpp
|
||||
Interface/Core/CodeCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/SharedCodeBufferManager.cpp
|
||||
Interface/Core/OpcodeDispatcher/AVX_128.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
@@ -68,6 +71,7 @@ set(SRCS
|
||||
Utils/LongJump.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/WorkQueueThread.cpp
|
||||
Utils/Profiler.cpp)
|
||||
|
||||
if (ARCHITECTURE_arm64)
|
||||
@@ -258,6 +262,10 @@ endfunction()
|
||||
# Build FEXCore_Base static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base PUBLIC ${LIBS})
|
||||
if (MINGW)
|
||||
target_link_libraries(FEXCore_Base PUBLIC ntdll)
|
||||
endif()
|
||||
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
@@ -300,6 +308,7 @@ add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
|
||||
if (ENABLE_FEX_ALLOCATOR)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_FEX_ALLOCATOR=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC rpmalloc)
|
||||
target_include_directories(JemallocLibs PRIVATE "${PROJECT_SOURCE_DIR}/include/")
|
||||
endif()
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
|
||||
|
||||
@@ -18,7 +18,7 @@ struct BitSet final {
|
||||
constexpr static size_t MinimumSize = sizeof(ElementType);
|
||||
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
|
||||
|
||||
ElementType* Memory;
|
||||
ElementType* Memory {};
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
@@ -33,14 +33,15 @@ struct BitSet final {
|
||||
FEXCore::Allocator::free(Memory);
|
||||
Memory = nullptr;
|
||||
}
|
||||
bool Get(T Element) {
|
||||
[[nodiscard]]
|
||||
bool Get(T Element) const {
|
||||
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
|
||||
}
|
||||
void Set(T Element) {
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void Clear(T Element) {
|
||||
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void MemClear(size_t Elements) {
|
||||
memset(Memory, 0, ToBytes(Elements));
|
||||
@@ -48,13 +49,15 @@ struct BitSet final {
|
||||
void MemSet(size_t Elements) {
|
||||
memset(Memory, 0xFF, ToBytes(Elements));
|
||||
}
|
||||
uint32_t ToBytes(size_t Elements) {
|
||||
return AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
[[nodiscard]]
|
||||
static size_t ToBytes(size_t Elements) {
|
||||
return AlignUp(Elements, MinimumSizeBits) / 8;
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
bool operator[](T Element) {
|
||||
[[nodiscard]]
|
||||
bool operator[](T Element) const {
|
||||
return Get(Element);
|
||||
}
|
||||
};
|
||||
@@ -62,35 +65,37 @@ struct BitSet final {
|
||||
template<typename T>
|
||||
struct BitSetView final {
|
||||
using ElementType = T;
|
||||
constexpr static size_t MinimumSize = sizeof(ElementType);
|
||||
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
|
||||
constexpr static size_t MinimumSize = BitSet<T>::MinimumSize;
|
||||
constexpr static size_t MinimumSizeBits = BitSet<T>::MinimumSizeBits;
|
||||
|
||||
ElementType* Memory;
|
||||
ElementType* Memory {};
|
||||
|
||||
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
bool Get(T Element) {
|
||||
[[nodiscard]]
|
||||
bool Get(T Element) const {
|
||||
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
|
||||
}
|
||||
void Set(T Element) {
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void Clear(T Element) {
|
||||
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void MemClear(size_t Elements) {
|
||||
memset(Memory, 0, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
memset(Memory, 0, BitSet<T>::ToBytes(Elements));
|
||||
}
|
||||
void MemSet(size_t Elements) {
|
||||
memset(Memory, 0xFF, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
memset(Memory, 0xFF, BitSet<T>::ToBytes(Elements));
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
bool operator[](T Element) {
|
||||
[[nodiscard]]
|
||||
bool operator[](T Element) const {
|
||||
return Get(Element);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -15,14 +15,9 @@ JITSymbols::~JITSymbols() {
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
void JITSymbols::InitFile(uint32_t ProcessPID) {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
#endif
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", ProcessPID);
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
|
||||
@@ -35,7 +35,7 @@ public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void InitFile(uint32_t ProcessPID);
|
||||
void RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
|
||||
#include <concepts>
|
||||
#include <string_view>
|
||||
#include <cstdlib>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
template<std::integral T>
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/StringConv.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Utils/Config.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
@@ -37,8 +38,16 @@ namespace detail {
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
constexpr static std::array<std::string_view, FEXCore::Config::ConfigOption::CONFIG_MAX> option_names = {
|
||||
#define OPT_BASE(type, group, enum, json, default) #json,
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
} // namespace detail
|
||||
|
||||
std::string_view GetConfigJSONName(FEXCore::Config::ConfigOption option) {
|
||||
return FEXCore::Config::detail::option_names[option];
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
PATH_DATA_DIR_GLOBAL,
|
||||
@@ -47,6 +56,7 @@ enum Paths {
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_CONFIG_TELEMETRY_FOLDER,
|
||||
PATH_CACHE_DIR,
|
||||
PATH_LAST,
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
@@ -63,6 +73,10 @@ void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetCacheDirectory(const std::string_view Path) {
|
||||
Paths[PATH_CACHE_DIR] = Path;
|
||||
}
|
||||
|
||||
const fextl::string& GetTelemetryDirectory() {
|
||||
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
|
||||
if (Path.empty()) {
|
||||
@@ -90,6 +104,10 @@ const fextl::string& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetCacheDirectory() {
|
||||
return Paths[PATH_CACHE_DIR];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
@@ -252,7 +270,7 @@ void Load() {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
static fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
@@ -501,4 +519,43 @@ void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArray
|
||||
}
|
||||
}
|
||||
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
|
||||
#define CONFIG_AFFECTSCODEGEN
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
fextl::string SerializeForCache() {
|
||||
fextl::string Config {};
|
||||
|
||||
auto append_string_triple = [](fextl::string& Config, std::string_view Key, ConfigOption Option, auto Value) {
|
||||
Config.append(Key);
|
||||
Config.append(1, '\0');
|
||||
Config.append(fextl::fmt::format("{}", FEXCore::ToUnderlying(Option)));
|
||||
Config.append(1, '\0');
|
||||
Config.append(fextl::fmt::format("{}", Value));
|
||||
Config.append(1, '\0');
|
||||
};
|
||||
|
||||
const auto SerializeValue = [&Config, append_string_triple]<typename T, ConfigOption Option>(auto ConfigVal, const auto Default) {
|
||||
if (!Config_AffectsCodeGen[FEXCore::ToUnderlying(Option)]) {
|
||||
// Skip everything that the config says doesn't affect codegen.
|
||||
return;
|
||||
}
|
||||
append_string_triple(Config, FEXCore::Config::GetConfigJSONName(Option), Option, ConfigVal());
|
||||
};
|
||||
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
SerializeValue.template operator()<type, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
|
||||
#define OPT_STR(group, enum, json, default) \
|
||||
SerializeValue.template operator()<fextl::string, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Unsupported.
|
||||
#define OPT_STRENUM(group, enum, json, default) // Unsupported.
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
return Config;
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool CheckConfigMatches(std::string_view Config) {
|
||||
// Serialize current config and just check if it matches.
|
||||
return SerializeForCache() == Config;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Config
|
||||
@@ -4,6 +4,7 @@
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
@@ -12,6 +13,7 @@
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
@@ -19,6 +21,7 @@
|
||||
"EnableCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Enable the code caching subsystem"
|
||||
]
|
||||
@@ -26,6 +29,7 @@
|
||||
"EnableLazyCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enable lazy loading of chunks in code caches"
|
||||
]
|
||||
@@ -33,6 +37,7 @@
|
||||
"EnableCodeCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enable expensive validation when loading code caches"
|
||||
]
|
||||
@@ -40,6 +45,8 @@
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
"AffectsCodeGen": "true",
|
||||
"Comment": "Technically affects codegen, but this is serialized elsewhere.",
|
||||
"Enums": {
|
||||
"ENABLESVE": "enablesve",
|
||||
"DISABLESVE": "disablesve",
|
||||
@@ -84,7 +91,11 @@
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a",
|
||||
"ENABLEMOPS": "enablemops",
|
||||
"DISABLEMOPS": "disablemops"
|
||||
"DISABLEMOPS": "disablemops",
|
||||
"ENABLEI8MM": "enablei8mm",
|
||||
"DISABLEI8MM": "disablei8mm",
|
||||
"ENABLEDOTPROD": "enabledotprod",
|
||||
"DISABLEDOTPROD": "disabledotprod"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -109,12 +120,15 @@
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it",
|
||||
"\t{enable,disable}i8mm: Will force enable or disable i8mm even if the host doesn't support it",
|
||||
"\t{enable,disable}dotprod: Will force enable or disable dotprod even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Scales the cycle counter on systems that have low frequencies."
|
||||
]
|
||||
@@ -122,22 +136,115 @@
|
||||
"HideHybrid": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Hides hybrid CPU core arrangement."
|
||||
]
|
||||
},
|
||||
"SoftwareRNG": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Emulates RDRAND and RDSEED in software when the host does not implement FEAT_RNG."
|
||||
]
|
||||
},
|
||||
"CPUFeatureRegisters": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but this is serialized in to HostFeatures.",
|
||||
"Desc": [
|
||||
"Allows overriding cpu feature flags for manual testing"
|
||||
]
|
||||
},
|
||||
"DiskCache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables disk caching for code blocks"
|
||||
]
|
||||
},
|
||||
"DiskCacheFileMapping": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Maps cache files for faster reading"
|
||||
]
|
||||
},
|
||||
"DiskCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Debug mode that does nothing but validate code hits"
|
||||
]
|
||||
},
|
||||
"DiskCacheRelocationFilter": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Don't cache blocks with relocations pointing outside of any known region"
|
||||
]
|
||||
},
|
||||
"DiskCacheAnonCaching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Attempt to cache anonymous code"
|
||||
]
|
||||
},
|
||||
"DiskCacheMemorySize": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"How much memory, if any, to use as a cache for the cache"
|
||||
]
|
||||
},
|
||||
"DiskCachePath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Optional base directory override for disk cache"
|
||||
]
|
||||
},
|
||||
"DiskCacheRODBNames": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Optional list of extra read-only disk cache DBs to consider"
|
||||
]
|
||||
},
|
||||
"DiskCacheMaxFileSize": {
|
||||
"Type": "uint64",
|
||||
"Default": "1073741824",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Size limit on the main cache file - index not included. Default 1G"
|
||||
]
|
||||
},
|
||||
"DiskCachePruneStaleEntries": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Allows FEX to prune stale disk cache entries on startup.",
|
||||
"Frees space by removing cache entries associated with old FEX versions."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
@@ -152,6 +259,7 @@
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
@@ -159,6 +267,7 @@
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
@@ -166,6 +275,7 @@
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
@@ -180,6 +290,7 @@
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
@@ -187,6 +298,7 @@
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
@@ -196,6 +308,7 @@
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
@@ -203,6 +316,7 @@
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
@@ -211,6 +325,7 @@
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
@@ -219,6 +334,7 @@
|
||||
"DynamicL1CacheIncreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "250",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
|
||||
"Lower numbers means more aggressive scaling upward to the maximum size.",
|
||||
@@ -230,6 +346,7 @@
|
||||
"DynamicL1CacheDecreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "50",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
|
||||
"The higher the number, the more aggressively it reduces the L1 cache size.",
|
||||
@@ -243,6 +360,7 @@
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
@@ -250,6 +368,7 @@
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
@@ -257,6 +376,7 @@
|
||||
"DumpIR": {
|
||||
"Type": "str",
|
||||
"Default": "no",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to dump the IR in to.",
|
||||
"[no, stdout, stderr, server, <Folder>]"
|
||||
@@ -265,6 +385,7 @@
|
||||
"PassManagerDumpIR": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::PassManagerDumpIR::OFF",
|
||||
"AffectsCodeGen": "false",
|
||||
"Enums": {
|
||||
"BEFOREOPT": "beforeopt",
|
||||
"AFTEROPT": "afteropt",
|
||||
@@ -283,6 +404,7 @@
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
@@ -290,6 +412,7 @@
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
@@ -297,6 +420,7 @@
|
||||
"GlobalJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name all JIT state as one symbol",
|
||||
"Useful for querying how much time is spent inside of the JIT",
|
||||
@@ -306,6 +430,7 @@
|
||||
"LibraryJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols grouped by library",
|
||||
"Useful for querying how much time is spent in each guest library",
|
||||
@@ -315,6 +440,7 @@
|
||||
"BlockJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols",
|
||||
"Useful for determining hot blocks of code",
|
||||
@@ -324,6 +450,7 @@
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
@@ -334,6 +461,7 @@
|
||||
"InjectLibSegFault": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Sets the environment variable LD_PRELOAD=libSegFault.so",
|
||||
"This allows the user to very easily enable libSegFault without dealing with environment variables",
|
||||
@@ -345,6 +473,7 @@
|
||||
"Disassemble": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::Disassemble::OFF",
|
||||
"AffectsCodeGen": "false",
|
||||
"Enums": {
|
||||
"DISPATCHER": "dispatcher",
|
||||
"BLOCKS": "blocks",
|
||||
@@ -361,6 +490,7 @@
|
||||
"X86Disassemble": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables x86/x86-64 guest disassembly output for compiled blocks.",
|
||||
"Requires FEX to be built with -DENABLE_ZYDIS=TRUE"
|
||||
@@ -369,6 +499,7 @@
|
||||
"ForceSVEWidth": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Allows overriding the SVE width in the vixl simulator.",
|
||||
"Useful as a debugging feature."
|
||||
@@ -377,6 +508,7 @@
|
||||
"DisableTelemetry": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Disables telemetry at runtime.",
|
||||
"Useful for CI instcountCI mostly"
|
||||
@@ -387,6 +519,7 @@
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
@@ -394,6 +527,7 @@
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stderr, server, <Filename>]"
|
||||
@@ -402,6 +536,7 @@
|
||||
"TelemetryDirectory": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/fex-emu/Telemetry/}"
|
||||
@@ -410,6 +545,7 @@
|
||||
"ProfileStats": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
@@ -418,6 +554,7 @@
|
||||
"EnableGpuvisProfiling": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables profiling when FEX was built with the gpuvis profiler backend."
|
||||
]
|
||||
@@ -427,6 +564,7 @@
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"AffectsCodeGen": "true",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
@@ -439,6 +577,7 @@
|
||||
"TSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Controls TSO IR ops.",
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
@@ -447,6 +586,7 @@
|
||||
"VectorTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
|
||||
]
|
||||
@@ -454,6 +594,7 @@
|
||||
"MemcpySetTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
|
||||
"Only affects REP MOVS and REP STOS instructions"
|
||||
@@ -462,6 +603,7 @@
|
||||
"HalfBarrierTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
@@ -470,6 +612,7 @@
|
||||
"StrictInProcessSplitLocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
@@ -478,6 +621,7 @@
|
||||
"KernelUnalignedAtomicBackpatching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
|
||||
]
|
||||
@@ -485,6 +629,7 @@
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
@@ -493,6 +638,7 @@
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
@@ -500,6 +646,7 @@
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
@@ -508,6 +655,7 @@
|
||||
"HideHypervisorBit": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Hides the hypervisor CPUID bit when set.",
|
||||
"Should only be used for applications that have issues with this set."
|
||||
@@ -516,6 +664,7 @@
|
||||
"StartupSleep": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Sleeps the process at startup for a duration of seconds.",
|
||||
"Useful if an application crashes too quickly to attach a debugger."
|
||||
@@ -524,6 +673,7 @@
|
||||
"StartupSleepProcName": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
@@ -531,6 +681,7 @@
|
||||
"MonoHacks": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
|
||||
]
|
||||
@@ -540,6 +691,7 @@
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
@@ -547,6 +699,7 @@
|
||||
"NeedsSeccomp": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
@@ -554,6 +707,7 @@
|
||||
"ExtendedVolatileMetadata": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
|
||||
"Limited in its use but can be handy.",
|
||||
@@ -578,15 +732,18 @@
|
||||
"Misc": {
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false"
|
||||
},
|
||||
"APP_FILENAME": {
|
||||
"Type": "str",
|
||||
"Default": ""
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false"
|
||||
},
|
||||
"APP_CONFIG_NAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"This is the application config name that has been loaded.",
|
||||
"This differs from APP_FILENAME in two ways",
|
||||
@@ -597,16 +754,29 @@
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but this is serialized elsewhere."
|
||||
},
|
||||
"DISABLE_VIXL_INDIRECT_RUNTIME_CALLS": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but only shows up in the test harness.",
|
||||
"Desc": [
|
||||
"This option is used for the InstructionCountCI so it can generate the same codegen between Arm64 hosts and vixl simulator hosts.",
|
||||
"Vixl simulator indirect runtime calls are a special hlt instruction with metadata after it. Effectively making a custom call instruction.",
|
||||
"With visual simulator calls disabled, the code generation would be the same as on a native Arm64 host, but running the code is broken."
|
||||
]
|
||||
},
|
||||
"CONFIG_VERSION": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "true",
|
||||
"Comment": [
|
||||
"Meta option that if config has ever changed definitions dramatically enough that we can rev the version.",
|
||||
"Be mindful that this will invalidate all caches!"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -55,4 +55,10 @@ FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionN
|
||||
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
|
||||
return Thread->CPUBackend->IsAddressInCodeBuffer(Address) || CodeCache.IsAddressInMappedCodeBuffer(Address);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::RequiresRelocatableConstants() const {
|
||||
// Support relocation when generating a cache or when generating reference code for validation
|
||||
return CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHEVALIDATION() || DiskCache.IsWritingDiskCache();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -4,6 +4,8 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/DiskCache.h"
|
||||
#include "Interface/Core/SharedCodeBufferManager.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -81,7 +83,6 @@ public:
|
||||
FEX_CONFIG_OPT(EnableLazyCodeCaching, ENABLELAZYCODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
fextl::unique_ptr<MappedCodeCacheFile> LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo&, uint64_t FileStartVA) override;
|
||||
@@ -120,17 +121,20 @@ public:
|
||||
* Note that FEX relocations are unrelated to ELF/PE relocations.
|
||||
*
|
||||
* @param GuestDelta Guest address offset to apply to RIP-relative data
|
||||
* @param RelocationOffset Offset to subtract from relocation target offsets
|
||||
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
|
||||
*
|
||||
* @return Returns true on success
|
||||
*/
|
||||
[[nodiscard]]
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations,
|
||||
uint32_t RelocationOffset, bool ForStorage);
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
|
||||
|
||||
// Same but on disk cache packed relocations
|
||||
[[nodiscard]]
|
||||
bool ApplyPackedCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
|
||||
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs);
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::SharedCodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
@@ -155,32 +159,32 @@ public:
|
||||
void SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - Thread = CreateThread();
|
||||
* - Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
* - Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = InitialStack;
|
||||
* - CTX->ExecuteThread(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread = CreateThread(NewState);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecuteThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - Thread = CreateThread(CopyOfThreadState);
|
||||
* - ExecuteThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread = CreateThread(NewState);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) override;
|
||||
FEXCore::Core::InternalThreadState* CreateThread(const FEXCore::Core::CPUState* NewThreadState) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
@@ -201,6 +205,8 @@ public:
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
virtual void InitDiskCache() override {}
|
||||
|
||||
CodeCache& GetCodeCache() override {
|
||||
return CodeCache;
|
||||
}
|
||||
@@ -242,11 +248,15 @@ public:
|
||||
}
|
||||
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
std::atomic<uint64_t>& GetMonoBackPatcherBlock() {
|
||||
return MonoBackpatcherBlock;
|
||||
}
|
||||
|
||||
// Manual debugging tooling which is useful for developers.
|
||||
struct TrackingEmpty {
|
||||
// RIP stepping handling
|
||||
virtual void AddSingleStepTarget(uint64_t GuestRIP) {}
|
||||
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RipEnd) {}
|
||||
virtual void AllTargetSingleStep() {}
|
||||
virtual void RemoveSingleStepTarget(uint64_t GuestRIP) {}
|
||||
virtual bool IsSingleStepTarget(uint64_t GuestRIP) {
|
||||
@@ -269,6 +279,10 @@ public:
|
||||
SingleStepTargets.emplace(GuestRIP);
|
||||
}
|
||||
|
||||
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RIPEnd) override {
|
||||
SingleStepRanges.emplace_back(Range {RIPBegin, RIPEnd});
|
||||
}
|
||||
|
||||
void RemoveSingleStepTarget(uint64_t GuestRIP) override {
|
||||
SingleStepTargets.erase(GuestRIP);
|
||||
}
|
||||
@@ -278,7 +292,7 @@ public:
|
||||
}
|
||||
|
||||
bool IsSingleStepTarget(uint64_t GuestRIP) override {
|
||||
return SingleStepEverything || SingleStepTargets.contains(GuestRIP);
|
||||
return SingleStepEverything || SingleStepTargets.contains(GuestRIP) || IsInRange(GuestRIP);
|
||||
}
|
||||
|
||||
void AddWriteWatchPoint(uint64_t Ptr) override {
|
||||
@@ -302,6 +316,14 @@ public:
|
||||
fextl::set<uint64_t> SingleStepTargets {};
|
||||
fextl::set<uint64_t> WatchWriteTargets {};
|
||||
fextl::set<uint64_t> WatchReadTargets {};
|
||||
struct Range {
|
||||
uint64_t Begin, End;
|
||||
};
|
||||
fextl::vector<Range> SingleStepRanges {};
|
||||
|
||||
bool IsInRange(uint64_t RIP) const {
|
||||
return std::ranges::any_of(SingleStepRanges, [RIP](const auto& range) { return RIP >= range.Begin && RIP <= range.End; });
|
||||
}
|
||||
|
||||
static bool ContainsRange(const fextl::set<uint64_t>& Set, uint64_t Ptr, size_t Size) {
|
||||
for (auto it = Set.lower_bound(Ptr); it != Set.end(); --it) {
|
||||
@@ -346,10 +368,16 @@ public:
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(SoftwareRNG, SOFTWARERNG);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
|
||||
} Config;
|
||||
|
||||
bool SoftwareRNGEnabled() const {
|
||||
return Config.SoftwareRNG() && HostRNGAvailable;
|
||||
}
|
||||
bool HostRNGAvailable {};
|
||||
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {};
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
@@ -361,6 +389,7 @@ public:
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
DiskCache::DiskCache DiskCache;
|
||||
CodeCache CodeCache;
|
||||
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
|
||||
|
||||
@@ -438,6 +467,8 @@ public:
|
||||
return Config.MonoHacks && MonoDetected;
|
||||
}
|
||||
|
||||
bool RequiresRelocatableConstants() const;
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
|
||||
@@ -360,6 +360,7 @@ namespace x32 {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr, size_t size)
|
||||
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
|
||||
, EmitterCTX {ctx}
|
||||
, SupportCodeRelocations {ctx->RequiresRelocatableConstants()}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
|
||||
#endif
|
||||
@@ -425,7 +426,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
NOPPad = false;
|
||||
} else if (Pad == PadType::AUTOPAD) {
|
||||
// Force NOP padding to ensure relocated constants always have enough encoding space available
|
||||
NOPPad = EnableCodeCaching;
|
||||
NOPPad = SupportCodeRelocations;
|
||||
}
|
||||
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
@@ -633,7 +634,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs) {
|
||||
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
@@ -649,7 +650,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
|
||||
if (SetFIZ) {
|
||||
if (Options.SetFIZ) {
|
||||
// Insert MXCSR.DAZ in to FIZ
|
||||
ldr(TmpReg2.W(), STATE.R(), offsetof(FEXCore::Core::CPUState, mxcsr));
|
||||
bfxil(ARMEmitter::Size::i64Bit, TmpReg, TmpReg2, 6, 1);
|
||||
@@ -659,7 +660,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
#endif
|
||||
|
||||
if (SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
|
||||
if (Options.SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
@@ -822,7 +823,7 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
}
|
||||
|
||||
FillSpecialRegs(TmpReg, TmpReg2, true, Options.FPRs);
|
||||
FillSpecialRegs(TmpReg, TmpReg2, {.SetFIZ = true, .SetPredRegs = Options.FPRs});
|
||||
|
||||
if (Options.FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
@@ -1059,6 +1060,7 @@ size_t Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, boo
|
||||
SpillStaticRegs(TmpReg, {
|
||||
.GPRSpillMask = PreserveSRAMask,
|
||||
.FPRSpillMask = PreserveSRAFPRMask,
|
||||
.FPRs = FPRs,
|
||||
});
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -1121,6 +1123,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
LOGMAN_THROW_A_FMT((CurrentOffset & 3) == 0, "Can't Align16B code that isn't 4-byte aligned!");
|
||||
for (uint64_t i = (-CurrentOffset & 0xF); i != 0; i -= 4) {
|
||||
nop();
|
||||
}
|
||||
|
||||
@@ -129,7 +129,21 @@ protected:
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
bool SupportCodeRelocations;
|
||||
|
||||
struct FillSpecialRegsOptions {
|
||||
// Whether or not to set the FPCR.FIZ (flush inputs to zero) bit in the FPCR to
|
||||
// the current value of the emulated MXCSR.DAZ bit.
|
||||
// Will only attempt to do so, even when set to true, if and only if the host system
|
||||
// supports FEAT_AFP.
|
||||
bool SetFIZ {};
|
||||
|
||||
// Whether or not FillSpecialRegs should load our SVE predicate temporaries
|
||||
// with certain canned values that accelerate some operations. Will (obviously)
|
||||
// not load predicates, even if set to true, on host systems that do not support SVE.
|
||||
bool SetPredRegs {};
|
||||
};
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
// Returning REG_INVALID if there was no mapping.
|
||||
@@ -308,8 +322,6 @@ protected:
|
||||
|
||||
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
|
||||
#endif
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -11,20 +11,12 @@
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_INCREMENTAL_U8_INDEX
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
|
||||
@@ -275,9 +267,9 @@ namespace CPU {
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(CodeBufferManager& CodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
|
||||
CPUBackend::CPUBackend(SharedCodeBufferManager& SharedCodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
|
||||
: ThreadState(ThreadState)
|
||||
, CodeBuffers(CodeBuffers) {
|
||||
, SharedCodeBuffers(SharedCodeBuffers) {
|
||||
|
||||
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
|
||||
|
||||
@@ -316,11 +308,11 @@ namespace CPU {
|
||||
|
||||
CPUBackend::~CPUBackend() = default;
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
|
||||
auto CPUBackend::AcquireNewSharedCodeBuffer() -> CodeBuffer* {
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
|
||||
CurrentCodeBuffer = SharedCodeBuffers.StartLargerCodeBuffer();
|
||||
|
||||
RegisterForSignalHandler(std::move(PrevCodeBuffer));
|
||||
return CurrentCodeBuffer.get();
|
||||
@@ -338,7 +330,7 @@ namespace CPU {
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
auto NewCodeBuffer = CodeBuffers.GetLatest();
|
||||
auto NewCodeBuffer = SharedCodeBuffers.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
|
||||
@@ -346,107 +338,17 @@ namespace CPU {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
|
||||
return *Buffer.LookupCache;
|
||||
}
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size)
|
||||
: AllocatedSize(Size) {
|
||||
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
|
||||
|
||||
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
|
||||
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<void*>(Ptr), Size, FEXCore::Allocator::THPControl::Enable);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
|
||||
}
|
||||
|
||||
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
//
|
||||
// MDWE prevents applications from creating RWX memory mappings.
|
||||
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
|
||||
//
|
||||
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
|
||||
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
|
||||
//
|
||||
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
|
||||
//
|
||||
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
|
||||
// -1: The kernel doesn't support MDWE
|
||||
// 0: MDWE is supported but disabled
|
||||
// >0: MDWE is enabled, hence prohibiting RWX mappings
|
||||
#ifndef PR_GET_MDWE
|
||||
#define PR_GET_MDWE 66
|
||||
#endif
|
||||
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
|
||||
if (MDWE != -1 && MDWE != 0) {
|
||||
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
|
||||
}
|
||||
#endif
|
||||
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
OnCodeBufferAllocated(Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
|
||||
if (!Latest) {
|
||||
if (FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
|
||||
// Start with a larger code buffer to avoid resizes that would discard
|
||||
// code loaded from caches
|
||||
AllocateNew(MAX_CODE_SIZE);
|
||||
} else {
|
||||
AllocateNew(INITIAL_CODE_SIZE);
|
||||
}
|
||||
}
|
||||
return Latest;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer() {
|
||||
if (!Latest) {
|
||||
// Allocate initial CodeBuffer and return it
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->AllocatedSize;
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.AllocatedSize - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
|
||||
const auto CheckCodeBuffer = [](const CodeBuffer& Buffer, uintptr_t Address) {
|
||||
const auto BufferPtr = reinterpret_cast<uintptr_t>(Buffer.GetBufferBase());
|
||||
const uintptr_t LastPageAddr = BufferPtr + Buffer.UsableSize();
|
||||
return (Address >= BufferPtr && Address < LastPageAddr);
|
||||
};
|
||||
|
||||
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
for (auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
for (const auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
if (CheckCodeBuffer(*Buffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -8,6 +8,8 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/SharedCodeBufferManager.h"
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -16,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
@@ -41,63 +44,10 @@ namespace CodeSerialize {
|
||||
struct GuestToHostMap;
|
||||
|
||||
namespace CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t AllocatedSize; // including guard page; see UsableSize()
|
||||
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
CodeBuffer(CodeBuffer&& oth) = delete;
|
||||
CodeBuffer& operator=(CodeBuffer&&) = delete;
|
||||
|
||||
~CodeBuffer();
|
||||
|
||||
/// Returns the number of bytes available for storing code
|
||||
size_t UsableSize() const {
|
||||
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
|
||||
*
|
||||
* The CodeBuffer is managed as a partially persistent data structure:
|
||||
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
|
||||
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
|
||||
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
|
||||
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
|
||||
*/
|
||||
class CodeBufferManager {
|
||||
public:
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
|
||||
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
|
||||
|
||||
// Write offset into the latest CodeBuffer
|
||||
std::size_t LatestOffset {};
|
||||
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
public:
|
||||
|
||||
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*);
|
||||
CPUBackend(SharedCodeBufferManager&, FEXCore::Core::InternalThreadState*);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
|
||||
@@ -107,6 +57,8 @@ namespace CPU {
|
||||
fextl::map<uint64_t, uint8_t*> EntryPoints;
|
||||
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
|
||||
size_t Size;
|
||||
// Offset of BlockBegin from the start of the CodeBuffer it lives in
|
||||
uint64_t HostCodeOffset;
|
||||
};
|
||||
|
||||
// Header that can live at the start of a JIT block.
|
||||
@@ -166,6 +118,10 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
virtual CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) {
|
||||
return {};
|
||||
}
|
||||
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
@@ -189,8 +145,9 @@ namespace CPU {
|
||||
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
// Acquires a new shared code buffer, setting `CurrentCodeBuffer` and returning a pointer to it.
|
||||
[[nodiscard]]
|
||||
CodeBuffer* GetEmptyCodeBuffer();
|
||||
CodeBuffer* AcquireNewSharedCodeBuffer();
|
||||
|
||||
// This is the code buffer containing the main code under execution by this thread.
|
||||
// CheckCodeBufferUpdate must be used before compiling new code.
|
||||
@@ -199,7 +156,7 @@ namespace CPU {
|
||||
// Old CodeBuffer generations required to be valid until returning from signal handlers
|
||||
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
|
||||
|
||||
CodeBufferManager& CodeBuffers;
|
||||
SharedCodeBufferManager& SharedCodeBuffers;
|
||||
|
||||
private:
|
||||
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
|
||||
|
||||
@@ -100,7 +100,7 @@ namespace ProductNames {
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
|
||||
uint32_t GetCPUID_Syscall() {
|
||||
static uint32_t GetCPUID_Syscall() {
|
||||
uint32_t CPU {};
|
||||
FHU::Syscalls::getcpu(&CPU, nullptr);
|
||||
return CPU;
|
||||
@@ -148,7 +148,7 @@ uint64_t GetCycleCounterFrequency() {
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint32_t GetCPUID_TPIDRRO() {
|
||||
static uint32_t GetCPUID_TPIDRRO() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], TPIDRRO_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
@@ -316,9 +316,8 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
// Walk our list of CPUMIDRs to find the most little core
|
||||
for (size_t j = LowestMIDRIdx; j < CPUMIDRs.size(); ++j) {
|
||||
auto& MIDROption = CPUMIDRs[i];
|
||||
const auto& MIDROption = CPUMIDRs[j];
|
||||
if ((MIDROption.Implementer == Implementer && MIDROption.Part == Part) || (MIDROption.Implementer == 0 && MIDROption.Part == 0)) {
|
||||
|
||||
LowestMIDRIdx = j;
|
||||
LowestMIDR = MIDR;
|
||||
break;
|
||||
@@ -459,6 +458,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(GetCPUID() << 24); // Local APIC ID
|
||||
|
||||
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
|
||||
|
||||
Res.ecx = (1 << 0) | // SSE3
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
@@ -489,13 +490,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(SupportsAVX() << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual 8086 mode enhancements
|
||||
(0 << 2) | // Debugging extensions
|
||||
(0 << 3) | // Page size extension
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extension
|
||||
(1 << 4) | // RDTSC supported
|
||||
(1 << 5) | // MSR supported
|
||||
(1 << 6) | // PAE
|
||||
@@ -649,7 +650,20 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
// AVX-VNNI is only advertised when the CPU supports I8MM or Dot Product.
|
||||
// Without these features the implementation is so slow that it is likely
|
||||
// to harm performance.
|
||||
const uint32_t SupportsAVXVNNI = SupportsAVX() && (CTX->HostFeatures.SupportsI8MM || CTX->HostFeatures.SupportsDotProd);
|
||||
|
||||
if (Leaf == 0) {
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDPID = 1;
|
||||
#else
|
||||
// RDPID under WIN32 is only supported if CPUIndex is available in TPIDRRO.
|
||||
const uint32_t SUPPORTS_RDPID = SupportsCPUIndexInTPIDRRO;
|
||||
#endif
|
||||
|
||||
// Disable Enhanced REP MOVS when TSO is enabled.
|
||||
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
|
||||
// This is due to LRCPC performance on Cortex being abysmal.
|
||||
@@ -658,40 +672,44 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
|
||||
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
|
||||
|
||||
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
|
||||
|
||||
// Number of subfunctions
|
||||
Res.eax = 0x0;
|
||||
Res.ebx = (1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(SupportsAVX() << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
(0 << 12) | // Intel resource directory technology Monitoring
|
||||
(1 << 13) | // Deprecates FPU CS and DS
|
||||
(0 << 14) | // Intel MPX
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // AVX512-F
|
||||
(0 << 17) | // AVX512-DQ
|
||||
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
|
||||
(1 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // AVX512-IFMA
|
||||
(0 << 22) | // PCOMMIT (deprecated?)
|
||||
(1 << 23) | // CLFLUSHOPT instruction
|
||||
(1 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // AVX512-PF
|
||||
(0 << 27) | // AVX512-ER
|
||||
(0 << 28) | // AVX512-CD
|
||||
(Features.SHA << 29) | // SHA instructions
|
||||
(0 << 30) | // AVX512-BW
|
||||
(0 << 31); // AVX512-VL
|
||||
// TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional
|
||||
// on AVX-VNNI support. We should revisit this if/when we add more to this leaf.
|
||||
Res.eax = SupportsAVXVNNI;
|
||||
Res.ebx = (1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(SupportsAVX() << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
(0 << 12) | // Intel resource directory technology Monitoring
|
||||
(1 << 13) | // Deprecates FPU CS and DS
|
||||
(0 << 14) | // Intel MPX
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // AVX512-F
|
||||
(0 << 17) | // AVX512-DQ
|
||||
(SupportsRAND << 18) | // RDSEED
|
||||
(1 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // AVX512-IFMA
|
||||
(0 << 22) | // PCOMMIT (deprecated?)
|
||||
(1 << 23) | // CLFLUSHOPT instruction
|
||||
(1 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // AVX512-PF
|
||||
(0 << 27) | // AVX512-ER
|
||||
(0 << 28) | // AVX512-CD
|
||||
(Features.SHA << 29) | // SHA instructions
|
||||
(0 << 30) | // AVX512-BW
|
||||
(0 << 31); // AVX512-VL
|
||||
|
||||
Res.ecx = (1 << 0) | // PREFETCHWT1
|
||||
(0 << 1) | // AVX512VBMI
|
||||
@@ -715,7 +733,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 19) | // MPX MAWAU
|
||||
(0 << 20) | // MPX MAWAU
|
||||
(0 << 21) | // MPX MAWAU
|
||||
(1 << 22) | // RDPID Read Processor ID
|
||||
(SUPPORTS_RDPID << 22) | // RDPID Read Processor ID
|
||||
(0 << 23) | // AES Key Locker
|
||||
(1 << 24) | // bus-lock-detect
|
||||
(0 << 25) | // CLDEMOTE
|
||||
@@ -759,38 +777,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 30) | // Arch capabilities - MSR module specific
|
||||
(0 << 31); // SSBD - Speculative Store Bypass Disable
|
||||
} else if (Leaf == 1) {
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(0U << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(SupportsAVXVNNI << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
|
||||
// Bits 4-31 currently reserved.
|
||||
Res.ebx = (0U << 0) | // PPIN
|
||||
@@ -1091,7 +1109,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(1 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
@@ -1341,7 +1359,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO != 0}
|
||||
, GetCPUID {GetCPUID_Syscall} {
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/crc32.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
@@ -50,6 +51,10 @@ MappedCodeCacheFile::~MappedCodeCacheFile() {
|
||||
if (!CodeBuffer.empty()) {
|
||||
FEXCore::Allocator::munmap(CodeBuffer.data(), CodeBuffer.size_bytes());
|
||||
}
|
||||
#elif defined(_M_ARM64EC)
|
||||
if (!CodeBuffer.empty()) {
|
||||
FEXCore::Allocator::VirtualFree(CodeBuffer.data(), CodeBuffer.size_bytes());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -107,16 +112,22 @@ fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::if
|
||||
break;
|
||||
}
|
||||
Ret[Info.ExternalFileId].Filename = std::move(Filename);
|
||||
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
|
||||
} else if ((Entry.FileId == SetExecutableFileId::Marker32.FileId && Entry.BlockOffset == SetExecutableFileId::Marker32.BlockOffset) ||
|
||||
(Entry.FileId == SetExecutableFileId::Marker64.FileId && Entry.BlockOffset == SetExecutableFileId::Marker64.BlockOffset)) {
|
||||
CodeMapFileId ExecutableFileId;
|
||||
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
Ret[ExecutableFileId].IsExecutable = true;
|
||||
Ret[ExecutableFileId].ExecutableBitness =
|
||||
(Entry.FileId == SetExecutableFileId::Marker32.FileId && Entry.BlockOffset == SetExecutableFileId::Marker32.BlockOffset) ? 32 : 64;
|
||||
} else {
|
||||
if (!Ret.contains(Entry.FileId)) {
|
||||
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
|
||||
if (Entry.FileId == 0xffff'ffff'ffff'ffff) {
|
||||
ERROR_AND_DIE_FMT("Malformed code map");
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
|
||||
}
|
||||
} else {
|
||||
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
|
||||
}
|
||||
@@ -222,8 +233,8 @@ void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInf
|
||||
AppendData(std::as_bytes(std::span {Data, TotalSize}));
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
|
||||
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
|
||||
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo, bool Is64Bit) {
|
||||
CodeMap::SetExecutableFileId Data {Is64Bit ? CodeMap::SetExecutableFileId::Marker64 : CodeMap::SetExecutableFileId::Marker32, FileInfo.FileId};
|
||||
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
|
||||
}
|
||||
|
||||
@@ -262,16 +273,6 @@ CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
|
||||
if (Filename.empty()) {
|
||||
return 0xffff'ffff'ffff'ffff;
|
||||
}
|
||||
|
||||
// For now, we just use the file path as an identifier.
|
||||
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
|
||||
return XXH3_64bits(Filename.data(), Filename.size());
|
||||
}
|
||||
|
||||
struct CodeCacheHeader {
|
||||
std::array<char, 4> Magic = ExpectedMagic;
|
||||
// Version history:
|
||||
@@ -304,7 +305,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
std::ranges::copy(GIT_HASH, header.FEXVersion);
|
||||
header.NumBlocks = LookupCache.BlockList.size();
|
||||
header.NumCodePages = LookupCache.CodePages.size();
|
||||
header.CodeBufferSize = FEXCore::AlignUp(CTX.LatestOffset, Utils::FEX_PAGE_SIZE);
|
||||
header.CodeBufferSize = FEXCore::AlignUp(CodeBuffer->AllocatedSpaceUsed(), Utils::FEX_PAGE_SIZE);
|
||||
header.NumRelocations = Relocations.size();
|
||||
header.SerializedBaseAddress = SerializedBaseAddress;
|
||||
::write(fd, &header, sizeof(header));
|
||||
@@ -327,7 +328,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
|
||||
Guest -= SourceBinary.FileStartVA;
|
||||
::write(fd, &Guest, sizeof(Guest));
|
||||
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
|
||||
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->GetBufferBase());
|
||||
::write(fd, &HostCode, sizeof(HostCode));
|
||||
uint64_t NumCodePages = Host->CodePages.size();
|
||||
::write(fd, &NumCodePages, sizeof(NumCodePages));
|
||||
@@ -351,8 +352,9 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
}
|
||||
|
||||
// Dump the host code (relocated for position-independent serialization)
|
||||
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, 0, true)) {
|
||||
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->GetBufferBase()),
|
||||
reinterpret_cast<std::byte*>(CodeBuffer->GetBufferBase()) + CodeBuffer->AllocatedSpaceUsed());
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
|
||||
return false;
|
||||
}
|
||||
@@ -395,7 +397,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
ERROR_AND_DIE_FMT("Failed to create cache load validation context");
|
||||
}
|
||||
|
||||
ValidationThread.reset(ValidationCTX->CreateThread(0, 0, nullptr));
|
||||
ValidationThread.reset(ValidationCTX->CreateThread(nullptr));
|
||||
|
||||
auto Frame = ValidationThread->CurrentFrame;
|
||||
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &ValidationGDT[0];
|
||||
@@ -416,11 +418,12 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
while (CachedCode.size_bytes() > NewCodeBuffer->UsableSize()) {
|
||||
ValidationCTX->ClearCodeCache(ValidationThread.get());
|
||||
NewCodeBuffer = ValidationCTX->GetLatest();
|
||||
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->AllocatedSize / 1024 / 1024);
|
||||
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->TotalAllocationSize() / 1024 / 1024);
|
||||
}
|
||||
|
||||
std::span<std::byte> CodeBufferRangeRef =
|
||||
std::as_writable_bytes(std::span {NewCodeBuffer->Ptr, NewCodeBuffer->Ptr + NewCodeBuffer->UsableSize()}).subspan(0, CachedCode.size_bytes());
|
||||
std::as_writable_bytes(std::span {NewCodeBuffer->GetBufferBase(), NewCodeBuffer->GetBufferBase() + NewCodeBuffer->UsableSize()})
|
||||
.subspan(0, CachedCode.size_bytes());
|
||||
|
||||
while (!GuestBlocks.empty()) {
|
||||
auto [CompiledBlocks, _, _2, _3, _4] = ValidationCTX->CompileCode(ValidationThread.get(), *GuestBlocks.begin(), 0 /* TODO: Set MaxInst? */);
|
||||
@@ -434,12 +437,12 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
NewRelocations.erase(std::remove_if(NewRelocations.begin(), NewRelocations.end(), [](const CPU::Relocation& Reloc) {
|
||||
return Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL && Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
}));
|
||||
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, 0, false);
|
||||
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, false);
|
||||
|
||||
if (ValidationCTX->LatestOffset <= CodeBufferRangeRef.size()) {
|
||||
if (NewCodeBuffer->AllocatedSpaceUsed() <= CodeBufferRangeRef.size()) {
|
||||
// Reference compilation produced fewer bytes than our cache, so validation is going to fail.
|
||||
// Make sure we don't output any garbage bytes though.
|
||||
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, ValidationCTX->LatestOffset);
|
||||
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, NewCodeBuffer->AllocatedSpaceUsed());
|
||||
}
|
||||
|
||||
auto [Mismatch, _] = std::mismatch(CodeBufferRangeRef.begin(), CodeBufferRangeRef.end(), CachedCode.begin());
|
||||
@@ -466,7 +469,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
if (tail->RIP >= Section.BeginVA && tail->RIP < Section.EndVA) {
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, _] =
|
||||
ValidationCTX->GenerateIR(ValidationThread.get(), tail->RIP, false, FEXCore::Config::Get_MAXINST());
|
||||
fextl::stringstream ss;
|
||||
fextl::ostringstream ss;
|
||||
FEXCore::IR::Dump(&ss, &*IRView);
|
||||
LogMan::Msg::EFmt("IR:\n{}", ss.str());
|
||||
} else {
|
||||
@@ -490,47 +493,153 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
|
||||
// Reset Context state for next validation
|
||||
ValidationThread->LookupCache->ClearCache(ValidationThread->LookupCache->AcquireWriteLock());
|
||||
ValidationCTX->LatestOffset = 0;
|
||||
NewCodeBuffer->Reset();
|
||||
|
||||
LogMan::Msg::IFmt(" successfully validated cache");
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, uint32_t RelocationOffset, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset >= RelocationOffset, "Invalid relocation offset");
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset - RelocationOffset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset - RelocationOffset);
|
||||
static inline void ApplySymbolLiteralRelocation(ContextImpl& CTX, const CPU::RelocNamedSymbolLiteral::NamedSymbol Symbol,
|
||||
uint64_t GuestEntry, CPU::Arm64Emitter& Emitter, bool ForStorage) {
|
||||
// Generate a literal so we can place it
|
||||
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Symbol);
|
||||
Emitter.dc64(Pointer);
|
||||
}
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
static inline bool
|
||||
ApplyThunkMoveRelocation(ContextImpl& CTX, const IR::SHA256Sum* Symbol, uint32_t RegisterIndex, CPU::Arm64Emitter& Emitter, bool ForStorage) {
|
||||
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(*Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
return true;
|
||||
}
|
||||
|
||||
static inline void ApplyRIPLiteralRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint64_t GuestEntry, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.dc64(GuestEntry + GuestRIP);
|
||||
}
|
||||
|
||||
static inline void
|
||||
ApplyRIPMoveRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint8_t RegisterIndex, uint64_t GuestEntry, CPU::Arm64Emitter& Emitter) {
|
||||
uint64_t Pointer = GuestRIP + GuestEntry;
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline int64_t ReadLiveGuestData(uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
uint64_t Raw = 0;
|
||||
memcpy(&Raw, reinterpret_cast<const void*>(SiteAddress), ValueSize);
|
||||
// manual sign-extension from guest live bytes
|
||||
if (ValueSize == 1) {
|
||||
return (int8_t)Raw;
|
||||
} else if (ValueSize == 2) {
|
||||
return (int16_t)Raw;
|
||||
} else if (ValueSize == 4) {
|
||||
return (int32_t)Raw;
|
||||
} else {
|
||||
return (int64_t)Raw;
|
||||
}
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableDataRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), ReadLiveGuestData(SiteAddress, ValueSize),
|
||||
CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableRIPLiteralRelocation(uint64_t SiteAddress, uint8_t ValueSize, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.dc64(SiteAddress + ValueSize + ReadLiveGuestData(SiteAddress, ValueSize));
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableRIPMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
const uint64_t Target = SiteAddress + ValueSize + ReadLiveGuestData(SiteAddress, ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableCRCMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
const uint64_t Target = FEXCore::Utils::crc32(reinterpret_cast<const uint8_t*>(SiteAddress), ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyPackedCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
|
||||
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (auto& Reloc : SmallRelocs) {
|
||||
LOGMAN_THROW_A_FMT(Reloc.Offset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Offset);
|
||||
switch ((CPU::RelocationTypes)Reloc.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
// Generate a literal so we can place it
|
||||
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
|
||||
Emitter.dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer,
|
||||
CPU::Arm64Emitter::PadType::DOPAD);
|
||||
ApplySymbolLiteralRelocation(CTX, (CPU::RelocNamedSymbolLiteral::NamedSymbol)Reloc.Named.Symbol, GuestEntry, Emitter, false);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
|
||||
ApplyRIPLiteralRelocation(CTX, Reloc.RIPLiteral.GuestRIP, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
ApplyRIPMoveRelocation(CTX, Reloc.RIPMove.GuestRIP, Reloc.RIPMove.RegisterIndex, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE: {
|
||||
ApplyPatchableDataRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
|
||||
ApplyPatchableRIPLiteralRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE: {
|
||||
ApplyPatchableRIPMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE: {
|
||||
ApplyPatchableCRCMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unknown packed relocation type {}", ToUnderlying((CPU::RelocationTypes)Reloc.Type));
|
||||
}
|
||||
}
|
||||
for (auto& Reloc : ThunkRelocs) {
|
||||
LOGMAN_THROW_A_FMT(Reloc.Offset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Offset);
|
||||
if (!ApplyThunkMoveRelocation(CTX, (const IR::SHA256Sum*)Reloc.SymbolHash, Reloc.RegisterIndex, Emitter, false)) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset);
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
ApplySymbolLiteralRelocation(CTX, Reloc.NamedSymbolLiteral.Symbol, GuestEntry, Emitter, ForStorage);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
if (!ApplyThunkMoveRelocation(CTX, &Reloc.NamedThunkMove.Symbol, Reloc.NamedThunkMove.RegisterIndex, Emitter, ForStorage)) {
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
ApplyRIPLiteralRelocation(CTX, Reloc.GuestRIP.GuestRIP, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
ApplyRIPMoveRelocation(CTX, Reloc.GuestRIP.GuestRIP, Reloc.GuestRIP.RegisterIndex, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -600,7 +709,16 @@ CodeCache::LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo& F
|
||||
return nullptr;
|
||||
}
|
||||
auto CodeBuffer = std::span {static_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
|
||||
#else
|
||||
#elif defined(_M_ARM64EC)
|
||||
// TODO: Implement lazy mapping on Windows
|
||||
// NOTE: The executed code must have MEM_EXTENDED_PARAMETER_EC_CODE set, so we can't operate on the mapped cache file directly
|
||||
void* CodeBufferAllocation = Allocator::VirtualAlloc(header.CodeBufferSize, true);
|
||||
if (!CodeBufferAllocation) {
|
||||
LogMan::Msg::EFmt("Failed to allocate code cache memory");
|
||||
return nullptr;
|
||||
}
|
||||
auto CodeBuffer = std::span {reinterpret_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
|
||||
#else // WoW64
|
||||
// TODO: Implement lazy mapping on Windows
|
||||
auto CodeBuffer = CodeDataInFile;
|
||||
#endif
|
||||
@@ -852,7 +970,7 @@ void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte
|
||||
auto StagingSpan = std::span {Staging, Size};
|
||||
for (size_t i = StartPage; i < EndPage; ++i) {
|
||||
auto PageRelocations = SpanPageRelocations(Code, i);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, StagingSpan, PageRelocations, static_cast<uint32_t>(StartOffset), false);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, StagingSpan, PageRelocations, false);
|
||||
Code.LoadedPages[i] = true;
|
||||
}
|
||||
|
||||
@@ -870,9 +988,12 @@ void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte
|
||||
Allocator::VirtualDontNeed(Code.CodeBufferInFile.data() + StartOffset, Size);
|
||||
#else
|
||||
// TODO: Implement lazy mapping on Windows
|
||||
#ifdef _M_ARM64EC
|
||||
memcpy(Code.CodeBuffer.data() + StartOffset, Code.CodeBufferInFile.data() + StartOffset, Size);
|
||||
#endif
|
||||
for (size_t i = StartPage; i < EndPage; ++i) {
|
||||
auto PageRelocations = SpanPageRelocations(Code, i);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, 0, false);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, false);
|
||||
Code.LoadedPages[i] = true;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -30,6 +30,7 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/crc32.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
@@ -89,7 +90,7 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
if (Config.BlockJITNaming() || Config.GlobalJITNaming() || Config.LibraryJITNaming()) {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
Symbols.InitFile(Features.ProcessPID);
|
||||
}
|
||||
|
||||
uint64_t FrequencyCounter = FEXCore::GetCycleCounterFrequency();
|
||||
@@ -103,6 +104,16 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
#ifndef _WIN32
|
||||
// Check if the kernel supports getrandom().
|
||||
uint64_t Probe {};
|
||||
HostRNGAvailable = FHU::Syscalls::getrandom(&Probe, sizeof(Probe), 0) == sizeof(Probe);
|
||||
#else
|
||||
HostRNGAvailable = true;
|
||||
#endif
|
||||
|
||||
DiskCache.Init(this);
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
@@ -342,17 +353,17 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
|
||||
}
|
||||
|
||||
bool ContextImpl::InitCore() {
|
||||
if (CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
|
||||
// Start with a larger code buffer to avoid resizes that would discard code
|
||||
StartMaximalCodeBuffer();
|
||||
}
|
||||
|
||||
// Initialize the CPU core signal handlers & DispatcherConfig
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
|
||||
|
||||
#if defined(_WIN32) && !defined(ARCHITECTURE_arm64ec)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
@@ -390,7 +401,7 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>(this);
|
||||
|
||||
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
|
||||
@@ -399,28 +410,20 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
Thread->PassManager->InsertRegisterAllocationPass(this);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
|
||||
// We finalize *after* the CPU backend is initialized, as the CPU backend will
|
||||
// provide necessary register information to the register allocation pass.
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) {
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(const FEXCore::Core::CPUState* NewThreadState) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
@@ -480,7 +483,7 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
|
||||
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->AllocatedSize);
|
||||
Symbols.RegisterJITSpace(Buffer->GetBufferBase(), Buffer->TotalAllocationSize());
|
||||
}
|
||||
|
||||
{
|
||||
@@ -505,11 +508,11 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
|
||||
FEXCore::File::File FD = FEXCore::File::File::GetStdERR();
|
||||
fextl::stringstream out;
|
||||
fextl::ostringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR);
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
}
|
||||
|
||||
bool ContextImpl::CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
return Thread.FrontendDecoder->CheckIfCacheable(Thread, reinterpret_cast<const uint8_t*>(GuestRIP), GuestRIP, MaxInst);
|
||||
@@ -526,6 +529,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
bool HasCustomIR {};
|
||||
|
||||
bool WantsDiskCachePatching = DiskCache.IsReadingDiskCache() || DiskCache.IsWritingDiskCache();
|
||||
|
||||
if (HasCustomIRHandlers.load(std::memory_order_relaxed)) {
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
@@ -538,18 +543,14 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
if (!HasCustomIR) {
|
||||
const uint8_t* GuestCode {};
|
||||
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
const auto* GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
Thread->FrontendDecoder->DecodeLoop(GuestCode);
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
|
||||
const auto* BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
const auto& CodeBlocks = BlockInfo->Blocks;
|
||||
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
auto CodeBlocks = &BlockInfo->Blocks;
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, &CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
|
||||
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
|
||||
|
||||
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
|
||||
@@ -563,11 +564,17 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
#endif
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
for (size_t j = 0; j < CodeBlocks.size(); ++j) {
|
||||
const auto& Block = CodeBlocks[j];
|
||||
|
||||
// Dispatch failures and invalid instructions terminate only the decoded
|
||||
// block that contains them. Other block targets in the same multiblock
|
||||
// compilation unit are independent entry paths.
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
|
||||
#ifdef ZYDIS_DISASSEMBLER
|
||||
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks->size() > 1) {
|
||||
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks.size() > 1) {
|
||||
LogMan::Msg::IFmt(" Block {} Entry={:#x} NumInsts={}", j, Block.Entry, Block.NumInstructions);
|
||||
}
|
||||
#endif
|
||||
@@ -575,7 +582,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
bool BlockInForceTSOValidRange = false;
|
||||
auto InstForceTSOIt = ForceTSOInstructions.end();
|
||||
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); It != ForceTSOInstructions.end() && *It < Block.Entry + Block.Size) {
|
||||
InstForceTSOIt = It;
|
||||
BlockInForceTSOValidRange = true;
|
||||
}
|
||||
@@ -584,18 +591,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
const uint64_t InstsInBlock = Block.NumInstructions;
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
@@ -639,9 +644,12 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
|
||||
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
std::array<uint8_t, 0x10> CodeOriginal;
|
||||
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto Value = FEXCore::Utils::crc32(ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CRC = WantsDiskCachePatching ?
|
||||
Thread->OpDispatcher->_PatchableGuestCRC(IR::OpSize::i64Bit, Value, (int64_t)ExistingCodePtr, DecodedInfo->InstSize) :
|
||||
Thread->OpDispatcher->Constant(Value);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CRC, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
|
||||
|
||||
@@ -651,8 +659,12 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
// Generate a relocatable entry for invalidation purposes.
|
||||
auto EntryToInvalidate = Thread->OpDispatcher->_EntrypointOffset(GPRSize, 0);
|
||||
auto NewRIP = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
// Invalidate and exit the function
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry(EntryToInvalidate, NewRIP);
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -747,10 +759,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
#endif
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
|
||||
IR::IREmitter* IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = Thread->OpDispatcher->ShouldDumpIR();
|
||||
@@ -790,6 +802,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
// OpDispatcher IR already released in this case.
|
||||
return {{}, nullptr, 0, 0, false};
|
||||
}
|
||||
@@ -803,6 +817,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
// Raced to compile, release the OpDispatcher IR.
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
.StartAddr = 0,
|
||||
@@ -821,6 +837,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return {
|
||||
.CompiledCode = std::move(CompiledCode),
|
||||
.DebugData = std::move(DebugData),
|
||||
@@ -857,6 +875,47 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, MaxInst);
|
||||
|
||||
std::optional<ExecutableFileSectionInfo> Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
std::optional<DiskCache::CodeHitData> Hit;
|
||||
std::optional<uint64_t> DiskCacheGuestCodeKey;
|
||||
{
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedDiskCacheLookupTime);
|
||||
Hit = DiskCache.Lookup(Thread, Region, GuestRIP, DiskCacheGuestCodeKey);
|
||||
if (Hit && !DiskCache.IsValidating()) {
|
||||
auto LoadedCode = Thread->CPUBackend->LoadCachedCode(Hit->HostCode);
|
||||
if (LoadedCode.BlockBegin) {
|
||||
for (auto& CodePage : Hit->GuestPages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, Hit->EntryPointRIPs, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Hit->EntryPointRIPs.size() == Hit->EntryPointHostOffsets.size(), "Mismatched Disk Cache entrypoint pairs!");
|
||||
|
||||
uintptr_t CachedHostCode = 0;
|
||||
for (size_t i = 0; i < Hit->EntryPointRIPs.size(); i++) {
|
||||
void* HostAddr = LoadedCode.BlockBegin + Hit->EntryPointHostOffsets[i];
|
||||
Thread->LookupCache->AddBlockMapping(Thread, Hit->EntryPointRIPs[i], Hit->GuestPages, HostAddr);
|
||||
if (Hit->EntryPointRIPs[i] == GuestRIP) {
|
||||
CachedHostCode = reinterpret_cast<uintptr_t>(HostAddr);
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(CachedHostCode != 0, "Couldn't find GuestRIP in Disk Cache entrypoints!");
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheHitCount, 1);
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return CachedHostCode;
|
||||
}
|
||||
}
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheMissCount, 1);
|
||||
}
|
||||
|
||||
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
|
||||
|
||||
@@ -869,6 +928,13 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return reinterpret_cast<uintptr_t>(CodePtr);
|
||||
}
|
||||
|
||||
if (DiskCacheGuestCodeKey && Hit && DiskCache.IsValidating()) {
|
||||
DiskCache.Validate(*DiskCacheGuestCodeKey, *Hit, CompiledCode, Region);
|
||||
}
|
||||
|
||||
// if this ever fires, we need to serialize the offset into disk cache
|
||||
LOGMAN_THROW_A_FMT(StartAddr == GuestRIP, "StartAddr offset from GuestRIP");
|
||||
|
||||
// The core managed to compile the code.
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = CompiledCode.BlockBegin;
|
||||
@@ -908,11 +974,6 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
@@ -928,19 +989,37 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
// Disk Cache
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
if (DiskCacheGuestCodeKey) {
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations;
|
||||
if (DebugData && DebugData->Relocations) {
|
||||
Relocations = *DebugData->Relocations;
|
||||
}
|
||||
std::span<const uint8_t> GuestCode = {reinterpret_cast<const uint8_t*>(StartAddr), Length};
|
||||
const Frontend::Decoder::DecodedBlockInformation* BlockInfo =
|
||||
NeedsAddGuestCodeRanges ? Thread->FrontendDecoder->GetDecodedBlockInfo() : nullptr;
|
||||
DiskCache.Store(Thread, Region, GuestRIP, *DiskCacheGuestCodeKey, GuestCode, CompiledCode, Relocations, BlockInfo);
|
||||
}
|
||||
|
||||
if (CodeMapWriter && Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
|
||||
}
|
||||
|
||||
if (CodeMapWriter) {
|
||||
auto Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
if (Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
// Clear any relocations that might have been generated
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
@@ -953,6 +1032,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, 1);
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, _] = CompileCode(Thread, GuestRIP, 1);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,258 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/WorkQueueThread.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <stdint.h>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace DiskCache {
|
||||
|
||||
namespace MesaFOZ {
|
||||
|
||||
#define FOSSILIZE_BLOB_HASH_LENGTH 40 /* SHA1 hexadecimal string length */
|
||||
|
||||
struct __attribute__((packed)) foz_payload_key {
|
||||
uint8_t bytes[FOSSILIZE_BLOB_HASH_LENGTH];
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) foz_payload_header {
|
||||
uint32_t payload_size;
|
||||
uint32_t format;
|
||||
uint32_t crc;
|
||||
uint32_t uncompressed_size;
|
||||
};
|
||||
|
||||
struct mesa_index_db_file_entry;
|
||||
|
||||
} // namespace MesaFOZ
|
||||
|
||||
class IndexedDB;
|
||||
|
||||
struct MemoryLRUKey {
|
||||
uint64_t LookupKey;
|
||||
XXH128_hash_t GuestHash;
|
||||
uint64_t GuestFootprint;
|
||||
uint32_t Size;
|
||||
};
|
||||
|
||||
struct IndexEntry {
|
||||
IndexedDB* DB;
|
||||
uint64_t Offset;
|
||||
uint32_t Size;
|
||||
uint32_t GuestSize;
|
||||
XXH128_hash_t GuestHash;
|
||||
fextl::shared_ptr<fextl::vector<uint8_t>> MemoryBlob;
|
||||
fextl::vector<uint32_t> GuestExtents;
|
||||
std::optional<fextl::list<MemoryLRUKey>::iterator> LRUEntry;
|
||||
};
|
||||
|
||||
struct IndexCacheHead {
|
||||
struct IndexEntry MainEntry;
|
||||
uint64_t MainEntryFootprint;
|
||||
fextl::unique_ptr<fextl::multimap<uint64_t, IndexEntry>> MoreEntries; // sorted by guest footprint
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) BlobFixedHeader {
|
||||
uint32_t GuestSize;
|
||||
uint32_t HostSize;
|
||||
uint32_t EntryPointCount;
|
||||
uint32_t SmallRelocCount;
|
||||
uint32_t ThunkRelocCount;
|
||||
XXH128_hash_t GuestHash;
|
||||
};
|
||||
|
||||
// packed struct for types 0, 2 and 3. type 1 is bigger and separate below
|
||||
struct __attribute__((packed)) BlobSmallRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t Type;
|
||||
union {
|
||||
struct __attribute__((packed)) {
|
||||
uint32_t Symbol;
|
||||
} Named;
|
||||
struct __attribute__((packed)) {
|
||||
uint64_t GuestRIP;
|
||||
} RIPLiteral;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint64_t GuestRIP;
|
||||
} RIPMove;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t ValueSize;
|
||||
uint32_t SiteOffset;
|
||||
} PatchableData;
|
||||
};
|
||||
};
|
||||
|
||||
// type 1, implicit
|
||||
struct __attribute__((packed)) BlobThunkRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t SymbolHash[32]; // sha256sum in the real RelocNamedThunkMove
|
||||
};
|
||||
|
||||
struct CodeHitData {
|
||||
fextl::vector<uint8_t> Blob;
|
||||
std::span<uint8_t> HostCode;
|
||||
std::span<const uint64_t> GuestPages;
|
||||
std::span<uint64_t> EntryPointRIPs;
|
||||
std::span<const uint32_t> EntryPointHostOffsets;
|
||||
|
||||
// the spans above point to memory owned by the Blob vec, so it's important this can't be copied
|
||||
CodeHitData() = default;
|
||||
CodeHitData(CodeHitData&&) = default;
|
||||
CodeHitData& operator=(CodeHitData&&) = default;
|
||||
CodeHitData(const CodeHitData&) = delete;
|
||||
CodeHitData& operator=(const CodeHitData&) = delete;
|
||||
};
|
||||
|
||||
using Index = fextl::robin_map<uint64_t, IndexCacheHead>;
|
||||
|
||||
class FOZFile {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheFileName, bool ReadOnly);
|
||||
bool Lock(uint32_t TimeoutMS) {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Lock(TimeoutMS);
|
||||
}
|
||||
bool Unlock() {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Unlock();
|
||||
}
|
||||
File::File::FileHandleType GetHandle() {
|
||||
return FD ? FD->GetHandle() : (File::File::FileHandleType)-1;
|
||||
}
|
||||
ssize_t Size();
|
||||
bool ReadAll(fextl::vector<uint8_t>& Out); // from first blob
|
||||
bool ReadBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool WriteBlob(const MesaFOZ::foz_payload_key& Key, std::span<const std::span<const uint8_t>> BlobChunks, uint64_t& OutBlobOffset);
|
||||
|
||||
private:
|
||||
static constexpr uint32_t OPEN_LOCK_TIMEOUT_MS = 100;
|
||||
|
||||
fextl::string FileName;
|
||||
fextl::unique_ptr<File::File> FD;
|
||||
bool ReadOnly = false;
|
||||
};
|
||||
|
||||
class IndexedDB {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
void PopulateIndex(Index& CacheIndex, bool& FoundMetadata);
|
||||
bool ReadCacheBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob,
|
||||
MesaFOZ::mesa_index_db_file_entry& IndexEntry, std::span<const uint8_t> IndexBlob);
|
||||
bool Full() const {
|
||||
return MaxSizeReached;
|
||||
}
|
||||
|
||||
private:
|
||||
// stores run on the Writer, so returning quick isn't as important
|
||||
static constexpr uint32_t STORE_LOCK_TIMEOUT_MS = 1000;
|
||||
static constexpr uint64_t BIG_MAPPING_SIZE = 1ULL << 33;
|
||||
|
||||
FOZFile CacheFOZ;
|
||||
uint8_t* CacheFileMapping = nullptr;
|
||||
std::atomic<uint64_t> CacheFileSize;
|
||||
FOZFile IndexFOZ;
|
||||
bool ReadOnly = false;
|
||||
bool MaxSizeReached = false;
|
||||
|
||||
FEX_CONFIG_OPT(MaxFileSize, DISKCACHEMAXFILESIZE);
|
||||
};
|
||||
|
||||
class DiskCache {
|
||||
public:
|
||||
void Init(FEXCore::Context::ContextImpl* CTX);
|
||||
|
||||
std::optional<CodeHitData> Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP,
|
||||
std::optional<uint64_t>& GuestCodeKey);
|
||||
void Validate(uint64_t GuestCodeKey, const CodeHitData& Hit, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::optional<ExecutableFileSectionInfo> Region);
|
||||
bool Store(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP, uint64_t GuestCodeKey,
|
||||
std::span<const uint8_t> GuestCode, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations, const Frontend::Decoder::DecodedBlockInformation* DecodedBlockInfo);
|
||||
|
||||
bool IsWritingDiskCache() const {
|
||||
return WritingDiskCache;
|
||||
}
|
||||
bool IsReadingDiskCache() const {
|
||||
return ReadingDiskCache;
|
||||
}
|
||||
bool IsValidating() const {
|
||||
return Validation;
|
||||
}
|
||||
|
||||
private:
|
||||
bool OpenCacheDB(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
uint64_t MakeLookupKey(Core::InternalThreadState* Thread, const uint64_t ModuleOffset, bool Writable, bool MonoBackpatcher);
|
||||
IndexEntry* LookupLocked(const uint64_t LookupKey, const XXH128_hash_t& GuestHash, const uint64_t GuestFootprint);
|
||||
|
||||
bool ReadingDiskCache {};
|
||||
bool WritingDiskCache {};
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
XXH128_hash_t BucketHash;
|
||||
fextl::vector<fextl::unique_ptr<IndexedDB>> ROCacheDBs;
|
||||
fextl::unique_ptr<IndexedDB> RWCacheDB;
|
||||
Index Index;
|
||||
std::mutex IndexLock;
|
||||
bool FoundMetadata = false;
|
||||
struct CacheStoreWorkItem;
|
||||
|
||||
struct PruneMemoryLRUWorkItem;
|
||||
std::atomic<uint64_t> MemoryLRUCurrentSize {};
|
||||
std::mutex MemoryLRULock;
|
||||
fextl::list<MemoryLRUKey> MemoryLRU;
|
||||
|
||||
// the Writer holds references to all this stuff above and needs to be last
|
||||
fextl::unique_ptr<WorkQueueThread> Writer;
|
||||
|
||||
FEX_CONFIG_OPT(EnableDiskCache, DISKCACHE);
|
||||
FEX_CONFIG_OPT(Validation, DISKCACHEVALIDATION);
|
||||
FEX_CONFIG_OPT(MapDiskCacheFiles, DISKCACHEFILEMAPPING);
|
||||
FEX_CONFIG_OPT(RelocationFilter, DISKCACHERELOCATIONFILTER);
|
||||
FEX_CONFIG_OPT(AnonCaching, DISKCACHEANONCACHING);
|
||||
FEX_CONFIG_OPT(BasePathOverride, DISKCACHEPATH);
|
||||
FEX_CONFIG_OPT(RODBNames, DISKCACHERODBNAMES);
|
||||
FEX_CONFIG_OPT(MemoryLRUMaxSize, DISKCACHEMEMORYSIZE);
|
||||
|
||||
uint64_t MemoryLRUEvictThreshold = MemoryLRUMaxSize / 25;
|
||||
};
|
||||
|
||||
static constexpr uint16_t AnonPrefixGuestBytes = 64;
|
||||
// The current version of the diskcache.
|
||||
// This must be changed any time codegen changes occur!
|
||||
// Be aware of the impact of changing this frequently!
|
||||
static constexpr uint16_t FormatVersion = 29;
|
||||
|
||||
static constexpr uint32_t LOOKUP_KEY_MAX_BUCKET_DEPTH = 500;
|
||||
|
||||
} // namespace DiskCache
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -121,7 +121,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
|
||||
FillSpecialRegs(TMP1, TMP2, false, true);
|
||||
FillSpecialRegs(TMP1, TMP2, {.SetFIZ = false, .SetPredRegs = true});
|
||||
|
||||
// As ARM64EC uses this as an entrypoint for both guest calls and host returns, opportunistically try to return
|
||||
// using the call-ret stack to avoid unbalancing it.
|
||||
@@ -276,8 +276,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
constexpr size_t InterruptPageOffset =
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
|
||||
if constexpr (InterruptPageOffset <= 32760) {
|
||||
str(ARMEmitter::XReg::zr, STATE, InterruptPageOffset);
|
||||
} else {
|
||||
// Need to use vector 128-bit store for this range.
|
||||
// Doesn't matter which register we use to store.
|
||||
str(ARMEmitter::QReg::q0, STATE, InterruptPageOffset);
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
@@ -357,7 +364,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileSingleStep);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP }
|
||||
}
|
||||
@@ -507,6 +514,64 @@ void Dispatcher::EmitDispatcher() {
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
// All dynamic and static registers are spilled coming in to this handler.
|
||||
// It's also the end of block and RIP might have changed, so we jump directly to the top of the loop.
|
||||
ThreadDispatchSyscallHandler = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Store in the state that we are in a syscall
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
// Syscall result is in any static register that the frontend desired.
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
});
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
// All dynamic and static registers are spilled coming in to this handler.
|
||||
// It's also the end of block and RIP might have changed, so we jump directly to the top of the loop.
|
||||
ThreadDispatchRemoveCodeEntry = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Arguments are already in x0, x1. Just jump to the handler.
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
});
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -2620,6 +2685,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
Ptrs.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Ptrs.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Ptrs.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Ptrs.ThreadDispatchSyscallHandler = ThreadDispatchSyscallHandler;
|
||||
Ptrs.ThreadDispatchRemoveCodeEntry = ThreadDispatchRemoveCodeEntry;
|
||||
Ptrs.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Ptrs.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Ptrs.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
|
||||
@@ -81,6 +81,8 @@ private:
|
||||
uint64_t AbsoluteLoopTopAddressEnterECFillSRA {};
|
||||
uint64_t ThreadPauseHandlerAddress {};
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA {};
|
||||
uint64_t ThreadDispatchSyscallHandler {};
|
||||
uint64_t ThreadDispatchRemoveCodeEntry {};
|
||||
uint64_t ExitFunctionLinkerAddress {};
|
||||
uint64_t SignalHandlerReturnAddress {};
|
||||
uint64_t SignalHandlerReturnAddressRT {};
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
@@ -69,7 +70,6 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
: Thread {Thread}
|
||||
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
|
||||
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
@@ -89,6 +89,11 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Check for wraparound
|
||||
if (Address + Size < Address) {
|
||||
return false;
|
||||
}
|
||||
|
||||
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
|
||||
ExecutableRangeBase = RangeInfo.Base;
|
||||
@@ -110,9 +115,8 @@ bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
std::optional<uint8_t> Byte = PeekByte(0);
|
||||
if (!Byte) {
|
||||
if (!Byte || InstructionSize == MAX_INST_SIZE) {
|
||||
HitNonExecutableRange = true;
|
||||
// Pretend we read 0, the main decode loop will see HitNonExecutableRange and rollback the instruction.
|
||||
return 0;
|
||||
@@ -137,6 +141,8 @@ std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
|
||||
|
||||
uint64_t Res = 0;
|
||||
uint64_t Address = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize);
|
||||
LastFieldReadOffset = (uint8_t)InstructionSize;
|
||||
LastFieldReadSize = Size;
|
||||
if (CheckRangeExecutable(Address, Size)) {
|
||||
std::memcpy(&Res, &InstStream.AdjustedInstStream[InstructionSize], Size);
|
||||
} else {
|
||||
@@ -211,6 +217,7 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1;
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
Operand->Data.SIB.PatchableDisp = false;
|
||||
|
||||
// Only called when ModRM.mod != 0b11
|
||||
struct Encodings {
|
||||
@@ -286,6 +293,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
// SIB
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
Operand->Data.SIB.PatchableDisp = false;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
{
|
||||
@@ -324,6 +332,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
auto [Literal, IsRelocation] = ReadData(4);
|
||||
Operand->Type = IsRelocation ? DecodedOperand::OpType::RIPRelativeRelocation : DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value = Literal;
|
||||
Operand->Data.RIPLiteral.PatchableDisp = false;
|
||||
} else {
|
||||
// Register-direct addressing
|
||||
Operand->Type = DecodedOperand::OpType::GPRDirect;
|
||||
@@ -339,6 +348,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
Operand->Type = IsRelocation ? DecodedOperand::OpType::GPRIndirectRelocation : DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->Data.GPRIndirect.Displacement = Literal;
|
||||
Operand->Data.GPRIndirect.PatchableDisp = false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -491,15 +501,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
|
||||
auto* CurrentDest = &DecodeInst->Dest;
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
|
||||
// Some instructions hardcode their destination as RAX
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR =
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
@@ -616,28 +618,13 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
|
||||
++CurrentSrc;
|
||||
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
|
||||
if (Bytes <= 8 && Bytes > 0) {
|
||||
auto [Literal, IsRelocation] = ReadData(Bytes);
|
||||
if (IsRelocation) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::LiteralRelocation;
|
||||
@@ -662,6 +649,11 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
} else {
|
||||
// All real x86 instructions have byte sizes that are 8-bytes or less.
|
||||
// Thunk instruction has an additional 32-byte SHA256 payload that needs to be accounted for.
|
||||
InstructionSize += Bytes;
|
||||
Bytes = 0;
|
||||
}
|
||||
|
||||
@@ -1091,13 +1083,21 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
if (ErrorDuringDecoding != DecodedBlockStatus::SUCCESS || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
const auto InstSize = DecodeInst->InstSize;
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
auto Result = ErrorDuringDecoding != DecodedBlockStatus::SUCCESS ? ErrorDuringDecoding :
|
||||
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
|
||||
HitNonExecutableRange ? DecodedBlockStatus::NOEXEC_INST :
|
||||
DecodedBlockStatus::BAD_RELOCATION;
|
||||
DecodeInst->InstSize = 0;
|
||||
return Result;
|
||||
|
||||
// A decode error can be caused by substituting zero for an inaccessible
|
||||
// instruction byte, so the instruction fetch fault takes priority.
|
||||
if (HitNonExecutableRange) {
|
||||
return InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
}
|
||||
|
||||
if (HitBadRelocation) {
|
||||
return DecodedBlockStatus::BAD_RELOCATION;
|
||||
}
|
||||
|
||||
return ErrorDuringDecoding;
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
@@ -1321,6 +1321,13 @@ void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
.BlockStatus = BlockIt->BlockStatus,
|
||||
};
|
||||
|
||||
if (BlockIt->DataMasks.size()) {
|
||||
auto MaskIt = std::lower_bound(BlockIt->DataMasks.begin(), BlockIt->DataMasks.end(), SplitAddr,
|
||||
[](const DataMask& Mask, uint64_t Addr) { return Mask.FieldAddress < Addr; });
|
||||
SplitBlock.DataMasks.assign(MaskIt, BlockIt->DataMasks.end());
|
||||
BlockIt->DataMasks.erase(MaskIt, BlockIt->DataMasks.end());
|
||||
}
|
||||
|
||||
BlockIt->Size = SplitOffset;
|
||||
BlockIt->NumInstructions = SplitIdx;
|
||||
|
||||
@@ -1345,7 +1352,8 @@ const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 && RIP >= VSyscall_Base && RIP < VSyscall_End) {
|
||||
if (BlockInfo.Is64BitMode && CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Linux && RIP >= VSyscall_Base &&
|
||||
RIP < VSyscall_End) {
|
||||
// VSyscall
|
||||
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
|
||||
// Offset 0: vgettimeofday
|
||||
@@ -1365,106 +1373,217 @@ const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _
|
||||
}
|
||||
|
||||
bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
DecodeInstructionsAtEntry(&Thread, InstStream, PC, MaxInst);
|
||||
SetupDecodeInstructionsAtEntry(&Thread, PC, MaxInst);
|
||||
DecodeLoop(InstStream);
|
||||
bool Uncacheable = HitBadRelocation;
|
||||
DelayedDisownBuffer();
|
||||
return !Uncacheable;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
|
||||
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
// cmp *, imm8 - seen varying in mono jitted code
|
||||
{
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
if ((DecodeInst->OPRaw == 0x80 || DecodeInst->OPRaw == 0x83) && ModRM.reg == 7 && LastFieldReadSize == 1) {
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
LiteralToPatch = &Src;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (LiteralToPatch && LiteralToPatch->Literal() != 0) {
|
||||
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, DataMaskType::MOV, LastFieldReadSize});
|
||||
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
|
||||
uint64_t TotalInstructions {};
|
||||
|
||||
SectionMinAddress = 0;
|
||||
SectionMaxAddress = ~0ULL;
|
||||
Relocations = nullptr;
|
||||
|
||||
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
|
||||
// If generating cache, attempt to load section bounds and relocations
|
||||
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
|
||||
SectionMinAddress = SectionInfo->FileStartVA;
|
||||
SectionMaxAddress = SectionInfo->EndVA;
|
||||
Relocations = &SectionInfo->FileInfo.Relocations;
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
|
||||
// Entry is a jump target
|
||||
BlocksToDecode = {PC};
|
||||
|
||||
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
BlockInfo.CodePages = {CurrentCodePage};
|
||||
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
if (LastFieldReadSize < 4) {
|
||||
return;
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
bool FinalInstruction {false};
|
||||
DataMaskType Type = DataMaskType::NONE;
|
||||
|
||||
while (!FinalInstruction && !BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
// imm32 or imm64 at the end type instructions
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_LITERAL_PATCHABLE) {
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
LiteralToPatch = &Src;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
if (DecodeInst->Dest.IsLiteral()) {
|
||||
LiteralToPatch = &DecodeInst->Dest;
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
auto BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
// heuristic: if it's a small value, assume it's more likely to be part of the code around it
|
||||
bool IsMOV = (DecodeInst->OPRaw >= 0xB8 && DecodeInst->OPRaw <= 0xBF) || DecodeInst->OPRaw == 0xC7;
|
||||
if (LiteralToPatch && IsMOV && LiteralToPatch->Literal() < 0x10000ULL) {
|
||||
LiteralToPatch = nullptr;
|
||||
}
|
||||
if (LiteralToPatch && LiteralToPatch->Literal() != 0) {
|
||||
Type = DataMaskType::MOV;
|
||||
}
|
||||
}
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
BlockIt->IsEntryPoint = EntryBlock;
|
||||
// anything that has a patchable disp32
|
||||
bool TryDisp = DecodeInst->Flags & X86Tables::DecodeFlags::FLAG_DECODED_MODRM;
|
||||
if (TryDisp) {
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
{
|
||||
// todo we could handle both imm + disp with more load tracking
|
||||
bool FoundLiteral = false;
|
||||
FEXCore::X86Tables::DecodedOperand* OpToPatch = nullptr;
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
FoundLiteral = true;
|
||||
break;
|
||||
}
|
||||
if (Src.IsRIPRelative() || Src.IsGPRIndirect() || Src.IsSIB()) {
|
||||
OpToPatch = &Src;
|
||||
}
|
||||
}
|
||||
if (DecodeInst->Dest.IsLiteral()) {
|
||||
FoundLiteral = true;
|
||||
}
|
||||
if (DecodeInst->Dest.IsRIPRelative() || DecodeInst->Dest.IsGPRIndirect() || DecodeInst->Dest.IsSIB()) {
|
||||
OpToPatch = &DecodeInst->Dest;
|
||||
}
|
||||
if (!FoundLiteral && OpToPatch) {
|
||||
if (DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch == &IR::OpDispatchBuilder::NOPOp) {
|
||||
// if it's a nop disp, only mask data out, no other action required
|
||||
Type = DataMaskType::NOP;
|
||||
} else if (OpToPatch->IsRIPRelative()) {
|
||||
Type = DataMaskType::DISP;
|
||||
OpToPatch->Data.RIPLiteral.PatchableDisp = true;
|
||||
OpToPatch->Data.RIPLiteral.DispOffset = LastFieldReadOffset;
|
||||
} else if (OpToPatch->IsGPRIndirect()) {
|
||||
// filter out disp8
|
||||
if (!(ModRM.mod == 1)) {
|
||||
Type = DataMaskType::DISP;
|
||||
OpToPatch->Data.GPRIndirect.PatchableDisp = true;
|
||||
OpToPatch->Data.GPRIndirect.DispOffset = LastFieldReadOffset;
|
||||
}
|
||||
} else if (OpToPatch->IsSIB()) {
|
||||
FEXCore::X86Tables::SIBDecoded SIB;
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
// two disp32 cases here
|
||||
if (ModRM.mod == 0b10 || (ModRM.mod == 0 && SIB.base == 0b101)) {
|
||||
Type = DataMaskType::DISP;
|
||||
OpToPatch->Data.SIB.PatchableDisp = true;
|
||||
OpToPatch->Data.SIB.DispOffset = LastFieldReadOffset;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// jmp/call branches that use a literal rip-relative offset
|
||||
// some of those may be inlined by multiblock and will be cleaned up at decode end
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral() && DecodeInst->Src[0].Literal() != 0) {
|
||||
LiteralToPatch = &DecodeInst->Src[0];
|
||||
Type = DataMaskType::BRANCH;
|
||||
}
|
||||
|
||||
uint64_t PCOffset = 0;
|
||||
uint64_t BlockStartOffset = DecodedSize;
|
||||
bool EraseBlock = true; // Unset once the block contains an instruction
|
||||
// todo add a bunch more
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
if (Type != DataMaskType::NONE) {
|
||||
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, Type, LastFieldReadSize});
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
if (LiteralToPatch) {
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Decoder::PruneInlinedBranchDataMasks() {
|
||||
for (auto& Block : BlockInfo.Blocks) {
|
||||
if (!Block.DataMasks.size()) {
|
||||
continue;
|
||||
}
|
||||
const auto& LastInst = Block.DecodedInstructions[Block.NumInstructions - 1];
|
||||
const auto& LastMask = Block.DataMasks.back();
|
||||
|
||||
if (LastMask.Type != DataMaskType::BRANCH) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const uint64_t NextInst = LastInst.PC + LastInst.InstSize;
|
||||
if (LastMask.FieldAddress < LastInst.PC || LastMask.FieldAddress + LastMask.ValueSize > NextInst) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (std::ranges::binary_search(BlockInfo.Blocks, NextInst + LastInst.Src[0].Data.LiteralPatchable.Value, std::less {}, &DecodedBlocks::Entry)) {
|
||||
Block.DataMasks.pop_back();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
// counter-intuitively, the masks are also needed for lookup on anon prefix decodes, not just stores
|
||||
bool WantsDataMasks = CTX->DiskCache.IsReadingDiskCache() || CTX->DiskCache.IsWritingDiskCache();
|
||||
|
||||
while (!FinalInstruction && (Paused || !BlocksToDecode.empty())) {
|
||||
bool Pausing = false;
|
||||
fextl::vector<DecodedBlocks>::iterator BlockIt;
|
||||
if (!Paused || BlockResume == -1) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress == ~0ULL || NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
BlockIt->IsEntryPoint = EntryBlock;
|
||||
|
||||
PCOffset = 0;
|
||||
BlockStartOffset = DecodedSize;
|
||||
EraseBlock = true; // Unset once the block contains an instruction
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
} else if (BlockResume != -1) {
|
||||
BlockIt = BlockInfo.Blocks.begin() + BlockResume;
|
||||
BlockResume = -1;
|
||||
}
|
||||
|
||||
Paused = false;
|
||||
|
||||
while (1) {
|
||||
InstructionSize = 0;
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpAddress = RIPToDecode + PCOffset;
|
||||
auto OpAddress = BlockIt->Entry + PCOffset;
|
||||
auto OpMaxAddress = OpAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
@@ -1486,6 +1605,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
BlockInfo.CodePages.insert(CurrentCodePage);
|
||||
}
|
||||
|
||||
LastFieldReadSize = 0;
|
||||
BlockIt->BlockStatus = DecodeInstruction(OpAddress);
|
||||
if (HitBadRelocation) {
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
@@ -1510,6 +1630,11 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
++BlockIt->NumInstructions;
|
||||
BlockIt->Size += DecodeInst->InstSize;
|
||||
|
||||
// if we weren't provided relocations (guest JIT), try to detect what we can
|
||||
if (WantsDataMasks && BlockIt->BlockStatus == DecodedBlockStatus::SUCCESS && BlockInfo.Is64BitMode && !Relocations) {
|
||||
DetectDataMasks(OpAddress, *BlockIt);
|
||||
}
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (BlockIt->BlockStatus != DecodedBlockStatus::SUCCESS) [[unlikely]] {
|
||||
if (!EntryBlock && BlockIt->BlockStatus != DecodedBlockStatus::BAD_RELOCATION) {
|
||||
@@ -1519,6 +1644,9 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
TotalInstructions -= BlockIt->NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
InstStream -= PCOffset;
|
||||
if (DecodedMaxAddress == OpEndAddress) {
|
||||
DecodedMaxAddress -= PCOffset;
|
||||
}
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
@@ -1532,6 +1660,15 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
break;
|
||||
}
|
||||
|
||||
if (GuestSizePause) {
|
||||
if (GuestSizePause > DecodeInst->InstSize) {
|
||||
GuestSizePause -= DecodeInst->InstSize;
|
||||
} else {
|
||||
GuestSizePause = 0;
|
||||
Pausing = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we need to end the entire multiblock
|
||||
FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (FinalInstruction) {
|
||||
@@ -1544,7 +1681,9 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
|
||||
if (CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions)) {
|
||||
BlockIt->ForceFullSMCDetection = true;
|
||||
}
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
@@ -1553,6 +1692,17 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
PCOffset += DecodeInst->InstSize;
|
||||
InstStream += DecodeInst->InstSize;
|
||||
|
||||
if (Pausing) {
|
||||
Pausing = false;
|
||||
Paused = true;
|
||||
BlockResume = BlockIt - BlockInfo.Blocks.begin();
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Paused) {
|
||||
break;
|
||||
}
|
||||
|
||||
// NOTE: BlockIt is only valid here in the EraseBlock case
|
||||
@@ -1564,6 +1714,16 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
CurrentBlockTargets.clear();
|
||||
EntryBlock = false;
|
||||
|
||||
if (Pausing && !BlocksToDecode.empty() && !FinalInstruction) {
|
||||
Paused = true;
|
||||
BlockResume = -1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Paused) {
|
||||
return;
|
||||
}
|
||||
|
||||
BlockInfo.TotalInstructionCount = TotalInstructions;
|
||||
@@ -1571,6 +1731,65 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
for (auto& Block : BlockInfo.Blocks) {
|
||||
Block.IsEntryPoint = BlockInfo.EntryPoints.contains(Block.Entry);
|
||||
}
|
||||
|
||||
// now that multiblock has settled down, remove any branch masks we put down that didn't end the block
|
||||
if (WantsDataMasks) {
|
||||
PruneInlinedBranchDataMasks();
|
||||
}
|
||||
}
|
||||
|
||||
void Decoder::SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
Paused = false;
|
||||
BlockResume = -1;
|
||||
DecodedSize = 0;
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
this->MaxInst = MaxInst;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
|
||||
TotalInstructions = 0;
|
||||
|
||||
SectionMinAddress = 0;
|
||||
SectionMaxAddress = ~0ULL;
|
||||
Relocations = nullptr;
|
||||
|
||||
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
|
||||
// If generating cache, attempt to load section bounds and relocations
|
||||
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
|
||||
SectionMinAddress = SectionInfo->FileStartVA;
|
||||
SectionMaxAddress = SectionInfo->EndVA;
|
||||
Relocations = &SectionInfo->FileInfo.Relocations;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
|
||||
// Entry is a jump target
|
||||
BlocksToDecode = {PC};
|
||||
|
||||
CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
BlockInfo.CodePages = {CurrentCodePage};
|
||||
|
||||
EntryBlock = true;
|
||||
FinalInstruction = false;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -19,9 +19,6 @@
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::HLE {
|
||||
enum class SyscallOSABI;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
class Decoder final {
|
||||
@@ -35,6 +32,14 @@ public:
|
||||
UNIMPLEMENTED_INST,
|
||||
};
|
||||
|
||||
enum class DataMaskType : uint8_t { NONE, MOV, BRANCH, DISP, NOP };
|
||||
|
||||
struct DataMask final {
|
||||
uint64_t FieldAddress;
|
||||
DataMaskType Type;
|
||||
uint8_t ValueSize;
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
struct DecodedBlocks final {
|
||||
uint64_t Entry {};
|
||||
@@ -44,6 +49,7 @@ public:
|
||||
DecodedBlockStatus BlockStatus;
|
||||
bool IsEntryPoint {};
|
||||
bool ForceFullSMCDetection {};
|
||||
fextl::vector<DataMask> DataMasks;
|
||||
};
|
||||
|
||||
struct DecodedBlockInformation final {
|
||||
@@ -57,7 +63,8 @@ public:
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
bool CheckIfCacheable(FEXCore::Core::InternalThreadState&, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
void SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeLoop(const uint8_t* InstStream, uint64_t GuestPause = 0);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
@@ -74,6 +81,10 @@ public:
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
PoolObject.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
void ResetExecutableRangeCache() {
|
||||
ExecutableRangeBase = ExecutableRangeEnd = 0;
|
||||
}
|
||||
@@ -89,7 +100,6 @@ private:
|
||||
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI {};
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
@@ -102,6 +112,9 @@ private:
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
|
||||
void DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block);
|
||||
void PruneInlinedBranchDataMasks();
|
||||
|
||||
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
|
||||
|
||||
uint8_t ReadByte();
|
||||
@@ -121,6 +134,19 @@ private:
|
||||
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
|
||||
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
uint64_t TotalInstructions {};
|
||||
uint64_t CurrentCodePage {};
|
||||
bool EntryBlock {};
|
||||
bool FinalInstruction {};
|
||||
uint64_t MaxInst {};
|
||||
bool Paused {};
|
||||
int64_t BlockResume = -1;
|
||||
uint64_t PCOffset {};
|
||||
uint64_t BlockStartOffset {};
|
||||
bool EraseBlock {};
|
||||
|
||||
uint8_t LastFieldReadOffset;
|
||||
uint8_t LastFieldReadSize;
|
||||
|
||||
uint64_t ExecutableRangeBase {};
|
||||
uint64_t ExecutableRangeEnd {};
|
||||
@@ -155,6 +181,7 @@ private:
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize {};
|
||||
// Contains the full decoded instruction, unless it is a `Thunk` instruction.
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
uint8_t LastEscapePrefix {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
@@ -10,11 +10,6 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_UNKNOWN, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint64_t* ABIHandlers) {
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = {ABIHandlers[FABI_F80_I16_F32_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4)};
|
||||
@@ -105,6 +100,7 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
|
||||
|
||||
// SSE4.2 string instructions
|
||||
// NOTE: Currently unused. See VectorFallbacks.h.
|
||||
Info[Core::OPINDEX_VPCMPESTRX] = {ABIHandlers[FABI_I32_I64_I64_V128_V128_I16],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle)};
|
||||
Info[Core::OPINDEX_VPCMPISTRX] = {ABIHandlers[FABI_I32_V128_V128_I16],
|
||||
@@ -216,12 +212,6 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include <cstring>
|
||||
|
||||
// NOTE: Currently unused. See VectorFallbacks.h.
|
||||
namespace FEXCore::CPU {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
|
||||
@@ -13,29 +13,33 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// PCMPXSTRX control byte fields
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
enum class SourceData {
|
||||
U8,
|
||||
U16,
|
||||
S8,
|
||||
S16,
|
||||
};
|
||||
|
||||
enum class Polarity {
|
||||
Positive,
|
||||
Negative,
|
||||
PositiveMasked,
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
// NB: The fallback handlers for the *PMCP*STR* instructions
|
||||
// are no longer used since we now emit inline ASM for them.
|
||||
// We preserve this as a reference implementation.
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
enum class SourceData {
|
||||
U8,
|
||||
U16,
|
||||
S8,
|
||||
S16,
|
||||
};
|
||||
|
||||
enum class Polarity {
|
||||
Positive,
|
||||
Negative,
|
||||
PositiveMasked,
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(uint64_t RAX, uint64_t RDX, VectorRegType lhs_v, VectorRegType rhs_v, uint16_t control) {
|
||||
__uint128_t lhs;
|
||||
memcpy(&lhs, &lhs_v, sizeof(lhs));
|
||||
|
||||
@@ -9,13 +9,11 @@ $end_info$
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
@@ -67,6 +65,21 @@ DEF_OP(EntrypointOffset) {
|
||||
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestData) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestData>();
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestRIP) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestRIP>();
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestCRC) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestRIP>();
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
// nop
|
||||
}
|
||||
@@ -421,8 +434,8 @@ DEF_OP(MulH) {
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
sxtw(TMP1, Src1.W());
|
||||
sxtw(TMP2, Src2.W());
|
||||
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
|
||||
ubfx(ARMEmitter::Size::i32Bit, Dst, Dst, 32, 32);
|
||||
mul(ARMEmitter::Size::i64Bit, Dst, TMP1, TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, Dst, Dst, 32, 32);
|
||||
} else {
|
||||
smulh(Dst.X(), Src1.X(), Src2.X());
|
||||
}
|
||||
@@ -774,7 +787,7 @@ DEF_OP(PDep) {
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
(void)cbz(EmitSize, OrigMask, &Done);
|
||||
(void)cbz(EmitSize, Mask, &Done);
|
||||
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
|
||||
@@ -58,7 +58,8 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
|
||||
switch (Lit.MoveABI.Header.Type) {
|
||||
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
|
||||
Lit.MoveABI.Header.Offset = GetCursorOffset();
|
||||
break;
|
||||
}
|
||||
@@ -102,6 +103,37 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
auto Arm64JITCore::InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize) -> NamedSymbolLiteralPair {
|
||||
return {
|
||||
.Lit = GuestRIP,
|
||||
.MoveABI =
|
||||
{
|
||||
.GuestPatchableData = {.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL,
|
||||
},
|
||||
.RegisterIndex = 0, // unused
|
||||
.ValueSize = ValueSize,
|
||||
// NOTE: Cache serialization will subtract the unit entry address later
|
||||
.SiteAddress = SiteAddress},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableMove(FEXCore::CPU::RelocationTypes Type, ARMEmitter::Register Reg, uint64_t Value,
|
||||
uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = Type};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
|
||||
// this might get patched on disk cache load
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
|
||||
// Rebase relocations to library base address
|
||||
for (auto& Relocation : Relocations) {
|
||||
|
||||
@@ -342,8 +342,8 @@ DEF_OP(TelemetrySetValue) {
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
|
||||
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP4, TMP3, TMP2);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP4, &LoopTop);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -83,7 +83,12 @@ DEF_OP(ExitFunction) {
|
||||
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
if (Op->PatchSiteAddress) {
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE, EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress,
|
||||
Op->PatchSiteSize);
|
||||
} else {
|
||||
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
}
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
@@ -173,6 +178,7 @@ DEF_OP(ExitFunction) {
|
||||
ARMEmitter::ForwardLabel TFUnset;
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
// todo do we need to account for cache patching here?
|
||||
InsertGuestRIPMove(TMP1, NewRIP);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
|
||||
@@ -180,7 +186,7 @@ DEF_OP(ExitFunction) {
|
||||
(void)Bind(&TFUnset);
|
||||
}
|
||||
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call, Op->PatchSiteAddress, Op->PatchSiteSize);
|
||||
(void)Bind(&l_CallReturn);
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
}
|
||||
@@ -277,74 +283,11 @@ DEF_OP(CondJump) {
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X0: SyscallHandler
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
|
||||
SpillStaticRegs(TMP1, {
|
||||
.GPRSpillMask = GPRSpillMask,
|
||||
.FPRSpillMask = FPRSpillMask,
|
||||
});
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
// SP supporting move
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
.GPRFillMask = GPRSpillMask,
|
||||
.FPRFillMask = FPRSpillMask,
|
||||
});
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
// Jump to the syscall dispatch handler. We won't return after this.
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadDispatchSyscallHandler));
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -379,74 +322,74 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
auto OldCode = Op->CodeOriginal.data();
|
||||
auto Base = GetReg(Op->Header.Args[0]).X();
|
||||
auto Base = GetReg(Op->Address).X();
|
||||
int len = Op->CodeLength;
|
||||
int Offset = 0;
|
||||
ARMEmitter::ForwardLabel Fail;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto CRC32Reg = GetReg(Op->crc);
|
||||
|
||||
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
|
||||
while (len >= Size) {
|
||||
LoadData();
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
len -= Size;
|
||||
Offset += Size;
|
||||
}
|
||||
};
|
||||
// Changes to TMP1
|
||||
auto WorkingReg = ARMEmitter::XReg::zr;
|
||||
auto BaseReg = TMP2;
|
||||
auto TmpDataReg = TMP3;
|
||||
mov(ARMEmitter::Size::i64Bit, BaseReg, Base);
|
||||
|
||||
EmitCheck(8, [&]() {
|
||||
ldr(TMP1, Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 8) {
|
||||
ldr<ARMEmitter::IndexType::POST>(TmpDataReg, BaseReg, 8);
|
||||
crc32x(TMP1, WorkingReg, TmpDataReg);
|
||||
len -= 8;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
EmitCheck(4, [&]() {
|
||||
ldr(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 4) {
|
||||
ldr<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 4);
|
||||
crc32w(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 4;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
EmitCheck(2, [&]() {
|
||||
ldrh(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 2) {
|
||||
ldrh<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 2);
|
||||
crc32h(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 2;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
EmitCheck(1, [&]() {
|
||||
ldrb(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 1) {
|
||||
ldrb<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 1);
|
||||
crc32b(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 1;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
sub(ARMEmitter::Size::i32Bit, Dst, TMP1, CRC32Reg);
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
cbz_OrRestart(ARMEmitter::Size::i32Bit, Dst, &End);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
auto Op = IROp->C<IR::IROp_ThreadRemoveCodeEntry>();
|
||||
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Store the new RIP to go to.
|
||||
str(GetReg(Op->NewRIP).X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
|
||||
// Move the entry to ABI before saving state.
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->EntryToInvalidate));
|
||||
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
// X1: RIPToInvalidate
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
|
||||
// TODO: Relocations don't seem to be wired up to this...?
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry, CPU::Arm64Emitter::PadType::AUTOPAD);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegs();
|
||||
// Jump to the invalidate dispatch handler. We won't return after this.
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadDispatchRemoveCodeEntry));
|
||||
br(ARMEmitter::XReg::x2);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
|
||||
@@ -292,8 +292,6 @@ DEF_OP(Vector_FToS) {
|
||||
frinti(SubEmitSize, Dst.Z(), Mask.Merging(), Vector.Z());
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
@@ -559,26 +557,30 @@ DEF_OP(Vector_F64ToI32) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// This has a known precision issue that isn't easily resolvable without throwing away performance.
|
||||
// Doing the conversion in multi-stage steps has an issue that you can lose precision in the f32->i32 step if your source was f64.
|
||||
// To get around this with ASIMD FEX needs to use fcvtzs (Scalar, Integer, to GPR) for each F64 to be directly converted to i32.
|
||||
// This is a very costly transform that the SVE path doesn't need to do since it supports f64->i32 directly.
|
||||
// If this precision issue is necessary then we can add an option for it in the future.
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
///< skip TowardsZero as fcvtzs below already truncates toward zero on its own
|
||||
auto CVTReg = Dst.Q();
|
||||
switch (Round) {
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Q(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
// Now narrow from f64 to f32.
|
||||
fcvtn(ARMEmitter::SubRegSize::i32Bit, Dst.Q(), Dst.Q());
|
||||
///< Convert f64 directly to i64
|
||||
fcvtzs(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), CVTReg);
|
||||
|
||||
///< Convert the two F32 integrals to real integers.
|
||||
fcvtzs(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
///< Saturating narrow i64 -> i32
|
||||
///
|
||||
///< The caller(Vector_CVT_Float_To_Int32Impl) only fixes up positive overflow:
|
||||
///< it tests MaxF > Src (MaxF = 2^31) and swaps in CVTMAX_I32 (0x80000000) where
|
||||
///< the test fails.
|
||||
///
|
||||
///< Sources below INT32_MIN are handled by sqxtn:
|
||||
///< ARM saturates to INT32_MIN, which is 0x80000000 the same value as
|
||||
///< x86's integer-indefinite value.
|
||||
sqxtn(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -324,24 +324,62 @@ DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
case 0b00000001:
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
|
||||
break;
|
||||
case 0b00010000:
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
|
||||
break;
|
||||
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: {
|
||||
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
|
||||
break;
|
||||
}
|
||||
case 0b00000001: {
|
||||
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src1.Z(), Src1.Z());
|
||||
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), VTMP1.Z(), Src2.Z());
|
||||
break;
|
||||
}
|
||||
case 0b00010000:
|
||||
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src2.Z(), Src2.Z());
|
||||
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), VTMP1.Z());
|
||||
break;
|
||||
case 0b00010001: {
|
||||
pmullt(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: {
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
|
||||
break;
|
||||
}
|
||||
case 0b00000001: {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
|
||||
break;
|
||||
}
|
||||
case 0b00010000: {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
|
||||
break;
|
||||
}
|
||||
case 0b00010001: {
|
||||
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -36,7 +36,14 @@ $end_info$
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <optional>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#ifdef _WIN32
|
||||
#include <atomic>
|
||||
#else
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
struct DivRem {
|
||||
@@ -63,8 +70,36 @@ LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
};
|
||||
}
|
||||
|
||||
static void
|
||||
PrintValue(uint64_t Value) {
|
||||
#ifndef _WIN32
|
||||
|
||||
static std::optional<uint64_t>
|
||||
RDRANDFallback(uint64_t Reseed) {
|
||||
uint64_t Value {};
|
||||
FHU::Syscalls::getrandom(&Value, sizeof(Value), 0);
|
||||
return Value;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
// Windows does not have an equivalent to getrandom() without dynamically linking
|
||||
// bcrypt et al. Since this fallback is just for compat it does not need to be
|
||||
// cryptographic, so instead we vendor SplitMix64 as a naive fallback.
|
||||
|
||||
// Reference implementation by Sebastiano Vigna, public domain (CC0)
|
||||
// https://prng.di.unimi.it/splitmix64.c
|
||||
static std::atomic<uint64_t>
|
||||
RNGState {static_cast<uint64_t>(__builtin_readcyclecounter())};
|
||||
|
||||
static std::optional<uint64_t> RDRANDFallback(uint64_t Reseed) {
|
||||
const uint64_t State = RNGState.load(std::memory_order_relaxed) + 0x9E3779B97F4A7C15ULL;
|
||||
RNGState.store(State, std::memory_order_relaxed);
|
||||
uint64_t Value = (State ^ (State >> 30)) * 0xBF58476D1CE4E5B9ULL;
|
||||
Value = (Value ^ (Value >> 27)) * 0x94D049BB133111EBULL;
|
||||
return Value ^ (Value >> 31);
|
||||
}
|
||||
#endif
|
||||
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
@@ -616,13 +651,13 @@ void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(*ctx, Thread)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128 != 0}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256 != 0}
|
||||
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES != 0}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP != 0}
|
||||
, CTX {ctx}
|
||||
, TempAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
, TempCodeBufferAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -630,7 +665,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
RAPass->SetNumPairRegs(PairRegisters);
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
@@ -664,28 +699,20 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
Ptrs.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
|
||||
Ptrs.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
Ptrs.RDRANDFallback = reinterpret_cast<uint64_t>(RDRANDFallback);
|
||||
}
|
||||
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
CurrentCodeBuffer = SharedCodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::Arm64JITCore::";
|
||||
EmitString(JITString);
|
||||
Align();
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->AllocatedSize);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
|
||||
auto CodeBuffer = AcquireNewSharedCodeBuffer();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -775,7 +802,7 @@ void Arm64JITCore::EmitTFCheck() {
|
||||
void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
// Used only for gdbserver at the moment
|
||||
constexpr size_t InterruptPageOffset =
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
|
||||
if constexpr (InterruptPageOffset <= 32760) {
|
||||
@@ -787,7 +814,7 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _WIN32
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
@@ -822,6 +849,33 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
|
||||
|
||||
CodeBuffer::CodeBufferAllocation Arm64JITCore::AllocateCodeBufferInSharedCache(size_t Size) {
|
||||
CodeBuffer::CodeBufferAllocation AllocatedInfo {};
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
// Bring CodeBuffer up to date
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// Attempt to allocate a buffer from the SharedCodeBuffers.
|
||||
while (AllocatedInfo.BufferAllocationOffset == nullptr) {
|
||||
AllocatedInfo = CurrentCodeBuffer->AtomicAllocateBuffer(Size);
|
||||
|
||||
if (AllocatedInfo.BufferAllocationOffset == nullptr) {
|
||||
// If it didn't fit then clear the buffer and try again.
|
||||
// This has the possibility of migrating the SharedCodeBuffer. See above in `Arm64JITCore::ClearCache()`
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return AllocatedInfo;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
@@ -843,7 +897,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
|
||||
case RestartOptions::Control::NeedsLargerJITSpace:
|
||||
// Get rid of the claimed buffer immediately, we can't fit in it at all.
|
||||
TempAllocator.UnclaimBuffer();
|
||||
TempCodeBufferAllocator.UnclaimBuffer();
|
||||
SSANodeMultiplier *= 2;
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
|
||||
@@ -864,7 +918,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
|
||||
// This minimizes lock contention of CodeBufferWriteMutex.
|
||||
auto TempCodeBufferInfo = TempAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
|
||||
auto TempCodeBufferInfo = TempCodeBufferAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
|
||||
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
|
||||
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
@@ -874,6 +928,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
LOGMAN_THROW_A_FMT(GetCursorOffset() == 0, "Needs to be zero");
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
@@ -909,7 +964,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
@@ -1001,9 +1055,15 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
dc64(0); // HostCode
|
||||
if (PendingJumpThunk.PatchSiteAddress) {
|
||||
// GuestRIP with an extra step
|
||||
PlaceNamedSymbolLiteral(
|
||||
InsertGuestPatchableRIPLiteral(PendingJumpThunk.GuestRIP, PendingJumpThunk.PatchSiteAddress, PendingJumpThunk.PatchSiteSize));
|
||||
} else {
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
}
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
BindOrRestart(&l_ExitLink);
|
||||
@@ -1063,69 +1123,47 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
|
||||
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
|
||||
Align();
|
||||
// Make sure code is 16B aligned on the tail.
|
||||
// Can't use Align16B here as vl64pair can cause non-4byte alignment.
|
||||
Align(16);
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
// Beginning of emission is guaranteed to be offset zero. So the code data size is just the current cursor offset.
|
||||
CodeData.Size = GetCursorOffset();
|
||||
|
||||
// Finalize and write block tail data
|
||||
JITBlockTail.Size = CodeData.Size;
|
||||
{
|
||||
auto PrevCur = GetCursorOffset();
|
||||
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
|
||||
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
|
||||
SetCursorOffset(PrevCur);
|
||||
|
||||
// Emitter buffer is no longer used, guard against misuse by setting to nullptr.
|
||||
SetBuffer(nullptr, 0);
|
||||
}
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
{
|
||||
auto CodeBufferLock = std::unique_lock {CodeBuffers.CodeBufferWriteMutex};
|
||||
LOGMAN_THROW_A_FMT(CodeData.Size % 16 == 0, "Needs to be 16B aligned!");
|
||||
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->AllocatedSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset);
|
||||
Align16B();
|
||||
if ((GetCursorOffset() + TempSize) > CurrentCodeBuffer->UsableSize()) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
auto AllocatedInfo = AllocateCodeBufferInSharedCache(CodeData.Size);
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
LOGMAN_THROW_A_FMT((reinterpret_cast<uintptr_t>(AllocatedInfo.BufferAllocationOffset) % 16) == 0, "Allocated buffer wasn't 16B "
|
||||
"aligned?");
|
||||
|
||||
// Adjust host addresses
|
||||
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
const auto Delta = AllocatedInfo.BufferAllocationOffset - CodeData.BlockBegin;
|
||||
CodeData.BlockBegin += Delta;
|
||||
for (auto& EntryPoint : CodeData.EntryPoints) {
|
||||
EntryPoint.second += Delta;
|
||||
}
|
||||
CodeBegin += Delta;
|
||||
|
||||
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
|
||||
Relocations[Idx].Header.Offset += CodeBuffers.LatestOffset;
|
||||
}
|
||||
CodeData.HostCodeOffset = CodeData.BlockBegin - CurrentCodeBuffer->GetBufferBase();
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
memcpy(AllocatedInfo.BufferAllocationOffset, TempCodeBuffer, CodeData.Size);
|
||||
}
|
||||
|
||||
TempAllocator.DelayedDisownBuffer();
|
||||
TempCodeBufferAllocator.DelayedDisownBuffer();
|
||||
|
||||
ClearICache(CodeBegin, CodeOnlySize);
|
||||
|
||||
@@ -1161,6 +1199,22 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
return std::move(CodeData);
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span<const uint8_t> HostBytes) {
|
||||
// we stored it aligned, better still be?
|
||||
LOGMAN_THROW_A_FMT(HostBytes.size() % 16 == 0, "Needs to be 16B aligned!");
|
||||
auto AllocatedInfo = AllocateCodeBufferInSharedCache(HostBytes.size());
|
||||
|
||||
uint8_t* Dest = AllocatedInfo.BufferAllocationOffset;
|
||||
memcpy(Dest, HostBytes.data(), HostBytes.size());
|
||||
ClearICache(Dest, HostBytes.size());
|
||||
|
||||
CPUBackend::CompiledCode Result;
|
||||
Result.BlockBegin = Dest;
|
||||
Result.Size = HostBytes.size();
|
||||
Result.HostCodeOffset = Dest - CurrentCodeBuffer->GetBufferBase();
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0) {
|
||||
return;
|
||||
|
||||
@@ -54,6 +54,9 @@ public:
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
void ClearRelocations() override {
|
||||
@@ -102,10 +105,12 @@ private:
|
||||
uint64_t CallerAddress;
|
||||
uint64_t GuestRIP;
|
||||
ARMEmitter::ForwardLabel Label;
|
||||
uint64_t PatchSiteAddress = 0;
|
||||
uint8_t PatchSiteSize = 0;
|
||||
};
|
||||
fextl::vector<PendingJumpThunk> PendingJumpThunks;
|
||||
|
||||
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
|
||||
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempCodeBufferAllocator;
|
||||
|
||||
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
|
||||
@@ -342,8 +347,8 @@ private:
|
||||
uint32_t End;
|
||||
};
|
||||
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call, uint64_t PatchSiteAddress = 0, uint8_t PatchSiteSize = 0) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}, PatchSiteAddress, PatchSiteSize});
|
||||
auto& Thunk = PendingJumpThunks.back();
|
||||
BindOrRestart(&Thunk.Label);
|
||||
if (Call) {
|
||||
@@ -526,8 +531,6 @@ private:
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
@@ -561,6 +564,9 @@ private:
|
||||
*/
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
|
||||
void InsertGuestPatchableMove(FEXCore::CPU::RelocationTypes Type, ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress,
|
||||
uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
@@ -580,6 +586,11 @@ private:
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Like InsertGuestRIPLiteral, but with patch information to recompute value from live guest bytes at cache load time
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
@@ -626,6 +637,8 @@ private:
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
[[nodiscard]] CodeBuffer::CodeBufferAllocation AllocateCodeBufferInSharedCache(size_t Size);
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
///< Unhandled handler
|
||||
|
||||
@@ -267,7 +267,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
ldr(Dst.Q(), TMP1, Op->BaseOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
|
||||
ldur(Dst.Q(), TMP1, Op->BaseOffset);
|
||||
ldur(Dst.Q(), TMP1);
|
||||
}
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
@@ -333,7 +333,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
str(Value.Q(), TMP1, Op->BaseOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
|
||||
stur(Value.Q(), TMP1, Op->BaseOffset);
|
||||
stur(Value.Q(), TMP1);
|
||||
}
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
@@ -2401,13 +2401,13 @@ DEF_OP(CacheLineClear) {
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
if (CTX->HostFeatures.DCacheSize() >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, CurrentWorkingReg);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
|
||||
CurrentWorkingReg = TMP1;
|
||||
}
|
||||
}
|
||||
@@ -2430,13 +2430,13 @@ DEF_OP(CacheLineClean) {
|
||||
|
||||
// Clean dcache only
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
if (CTX->HostFeatures.DCacheSize() >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, CurrentWorkingReg);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
|
||||
CurrentWorkingReg = TMP1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -168,7 +168,7 @@ DEF_OP(PushRoundingMode) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
|
||||
}
|
||||
|
||||
@@ -282,7 +282,7 @@ DEF_OP(ProcessorID) {
|
||||
// Load the values returned by the kernel
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::WReg::w0, ARMEmitter::WReg::w1, ARMEmitter::Reg::rsp);
|
||||
// Deallocate stack space
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
@@ -309,7 +309,39 @@ DEF_OP(ProcessorID) {
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
|
||||
if (CTX->HostFeatures.SupportsRAND) {
|
||||
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
|
||||
return;
|
||||
}
|
||||
|
||||
// Software fallback, call the host RNG generator.
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// x0 = Reseed
|
||||
// x1 = Generator
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Op->GetReseeded ? 1 : 0);
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.RDRANDFallback));
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, x1
|
||||
// std::optional<uint64_t>: value in x0, engaged flag in the low byte of x1. Match the hardware behaviour of setting Z when
|
||||
// no number was produced.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1);
|
||||
tst(ARMEmitter::Size::i64Bit, TMP2, 0xFF);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
|
||||
@@ -25,6 +25,22 @@ enum class RelocationTypes : uint32_t {
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocGuestRIP
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
|
||||
// The frontend flagged those regions as patchable by the disk cache
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_DATA_MOVE,
|
||||
|
||||
// Same as GuestRipLiteral but patchable
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_LITERAL,
|
||||
|
||||
// Like PATCHABLE_RIP_LITERAL but puts it in a register
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_MOVE,
|
||||
|
||||
// Patchable guest CRC
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_CRC_MOVE,
|
||||
};
|
||||
|
||||
struct FEX_PACKED RelocationHeader final {
|
||||
@@ -73,6 +89,20 @@ struct RelocGuestRIP final {
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
struct RelocGuestPatchableData final {
|
||||
RelocationHeader Header {};
|
||||
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
uint8_t ValueSize;
|
||||
|
||||
char Pad[2];
|
||||
|
||||
uint64_t SiteAddress;
|
||||
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
// Clang 16 Can't default-initialize this union
|
||||
static Relocation Default() {
|
||||
@@ -93,6 +123,8 @@ union Relocation {
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIP GuestRIP;
|
||||
|
||||
RelocGuestPatchableData GuestPatchableData;
|
||||
};
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
|
||||
|
||||
@@ -1112,8 +1112,10 @@ DEF_OP(VAddP) {
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Extract the upper half first, since Dst may alias the LHS.
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
@@ -1139,9 +1141,16 @@ DEF_OP(VOrn) {
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
not_(ARMEmitter::SubRegSize::i8Bit, VTMP1.Z(), Pred, Vector2.Z());
|
||||
orr(Dst.Z(), Vector1.Z(), VTMP1.Z());
|
||||
if (Dst == Vector1) {
|
||||
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Dst.Z());
|
||||
} else if (Dst == Vector2) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
not_(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Pred, Dst.Z());
|
||||
orr(Dst.Z(), Vector1.Z(), Dst.Z());
|
||||
} else {
|
||||
movprfx(Dst.Z(), Vector1.Z());
|
||||
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Vector1.Z());
|
||||
}
|
||||
} else if (Is128Bit) {
|
||||
orn(Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
@@ -1165,8 +1174,7 @@ DEF_OP(VFAddV) {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
|
||||
}
|
||||
if (HostSupportsSVE128) {
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Pred = PRED_TMP_16B.Merging();
|
||||
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
|
||||
} else {
|
||||
@@ -1193,20 +1201,16 @@ DEF_OP(VAddV) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// SVE doesn't have an equivalent ADDV instruction, so we make do
|
||||
// by performing two Adv. SIMD ADDV operations on the high and low
|
||||
// 128-bit lanes and then sum them up.
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto CompactPred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Select all our upper elements to run ADDV over them.
|
||||
not_(CompactPred, Mask, PRED_TMP_16B);
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, Vector.Z());
|
||||
|
||||
addv(SubRegSize.Vector, VTMP2.Q(), Vector.Q());
|
||||
addv(SubRegSize.Vector, VTMP1.Q(), VTMP1.Q());
|
||||
add(SubRegSize.Vector, Dst.Q(), VTMP1.Q(), VTMP2.Q());
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
uaddv(SubRegSize.Vector, Dst.D(), Mask, Vector.Z());
|
||||
} else {
|
||||
const auto Mask = ARMEmitter::PReg::p0;
|
||||
uaddv(SubRegSize.Vector, VTMP1.D(), Mask, Vector.Z());
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), 0);
|
||||
ptrue(SubRegSize.Vector, Mask, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Mask.Merging(), VTMP1.Z());
|
||||
}
|
||||
} else {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
addp(SubRegSize.Scalar, Dst, Vector);
|
||||
@@ -1295,6 +1299,7 @@ DEF_OP(VFAddP) {
|
||||
const auto Op = IROp->C<IR::IROp_VFAddP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto IsScalar = OpSize == IR::OpSize::i64Bit;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1325,6 +1330,8 @@ DEF_OP(VFAddP) {
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
} else if (IsScalar) {
|
||||
faddp(SubRegSize, Dst.D(), VectorLower.D(), VectorUpper.D());
|
||||
} else {
|
||||
faddp(SubRegSize, Dst.Q(), VectorLower.Q(), VectorUpper.Q());
|
||||
}
|
||||
@@ -2162,6 +2169,46 @@ DEF_OP(VCMPGT) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUCMPGT) {
|
||||
const auto Op = IROp->C<IR::IROp_VUCMPGT>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// General idea is to compare for unsigned greater-than, bitwise NOT
|
||||
// the valid values, then ORR the NOTed values with the original
|
||||
// values to form entries that are all 1s.
|
||||
cmphi(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmhi(SubRegSize.Scalar, Dst, Vector1, Vector2);
|
||||
} else {
|
||||
cmhi(SubRegSize.Vector, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VCMPGTZ) {
|
||||
const auto Op = IROp->C<IR::IROp_VCMPGTZ>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -2785,17 +2832,17 @@ DEF_OP(VUShrSWide) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
lsr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
@@ -2851,17 +2898,17 @@ DEF_OP(VSShrSWide) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
asr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
@@ -2917,17 +2964,17 @@ DEF_OP(VUShlSWide) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
lsl_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
@@ -3175,19 +3222,12 @@ DEF_OP(VUShrI) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
} else {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (BitShift == 0) {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), Vector.Z());
|
||||
}
|
||||
} else {
|
||||
// SVE LSR is destructive, so lets set up the destination if
|
||||
// Vector doesn't already alias it.
|
||||
if (Dst != Vector) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
|
||||
lsr(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
}
|
||||
} else {
|
||||
if (BitShift == 0) {
|
||||
@@ -3201,48 +3241,6 @@ DEF_OP(VUShrI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUShraI) {
|
||||
const auto Op = IROp->C<IR::IROp_VUShraI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst == DestVector) {
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
} else {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), DestVector.Z());
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
} else {
|
||||
mov(VTMP1.Z(), DestVector.Z());
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (Dst == DestVector) {
|
||||
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
|
||||
} else {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Q(), DestVector.Q());
|
||||
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
|
||||
} else {
|
||||
mov(VTMP1.Q(), DestVector.Q());
|
||||
usra(SubRegSize, VTMP1.Q(), Vector.Q(), BitShift);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSShrI) {
|
||||
const auto Op = IROp->C<IR::IROp_VSShrI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -3258,19 +3256,12 @@ DEF_OP(VSShrI) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Shift == 0) {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), Vector.Z());
|
||||
}
|
||||
} else {
|
||||
// SVE ASR is destructive, so lets set up the destination if
|
||||
// Vector doesn't already alias it.
|
||||
if (Dst != Vector) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), Shift);
|
||||
asr(SubRegSize, Dst.Z(), Vector.Z(), Shift);
|
||||
}
|
||||
} else {
|
||||
if (Shift == 0) {
|
||||
@@ -3300,19 +3291,12 @@ DEF_OP(VShlI) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
} else {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (BitShift == 0) {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), Vector.Z());
|
||||
}
|
||||
} else {
|
||||
// SVE LSL is destructive, so lets set up the destination if
|
||||
// Vector doesn't already alias it.
|
||||
if (Dst != Vector) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
|
||||
lsl(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
}
|
||||
} else {
|
||||
if (BitShift == 0) {
|
||||
@@ -3339,8 +3323,13 @@ DEF_OP(VUShrNI) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
if (BitShift == 0) {
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
}
|
||||
} else {
|
||||
if (BitShift == 0) {
|
||||
xtn(SubRegSize, Dst.D(), Vector.D());
|
||||
@@ -3388,6 +3377,55 @@ DEF_OP(VUShrNI2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRSHRN) {
|
||||
const auto Op = IROp->C<IR::IROp_VRSHRN>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
rshrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
} else {
|
||||
rshrn(SubRegSize, Dst.D(), Vector.D(), BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRSHRNPair) {
|
||||
const auto Op = IROp->C<IR::IROp_VRSHRNPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
rshrnb(SubRegSize, VTMP1.Z(), VectorLower.Z(), BitShift);
|
||||
rshrnb(SubRegSize, VTMP2.Z(), VectorUpper.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
} else {
|
||||
if (Dst == VectorUpper) {
|
||||
// RSHRN writes the lower half and would destroy the upper input.
|
||||
mov(VTMP1.Q(), VectorUpper.Q());
|
||||
VectorUpper = VTMP1;
|
||||
}
|
||||
|
||||
rshrn(SubRegSize, Dst.D(), VectorLower.D(), BitShift);
|
||||
rshrn2(SubRegSize, Dst.Q(), VectorUpper.Q(), BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSXTL) {
|
||||
const auto Op = IROp->C<IR::IROp_VSXTL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -3593,9 +3631,13 @@ DEF_OP(VSQXTN2) {
|
||||
mov(Dst.Q(), VectorLower.Q());
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 1, VTMP2, 0);
|
||||
} else {
|
||||
mov(VTMP1.Q(), VectorLower.Q());
|
||||
sqxtn2(SubRegSize, VTMP1, VectorUpper);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
if (Dst == VectorLower) {
|
||||
sqxtn2(SubRegSize, VectorLower, VectorUpper);
|
||||
} else {
|
||||
mov(VTMP1.Q(), VectorLower.Q());
|
||||
sqxtn2(SubRegSize, VTMP1, VectorUpper);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3833,6 +3875,171 @@ DEF_OP(VMul) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUSDot) {
|
||||
///< Dest = Acc + dot(Vector1 (unsigned 8-bit), Vector2 (signed 8-bit))
|
||||
// Matches:
|
||||
// - SVE - USDOT
|
||||
// - ASIMD - USDOT
|
||||
const auto Op = IROp->C<IR::IROp_VUSDot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// USDOT accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
DestTmp = Dst;
|
||||
} else {
|
||||
DestTmp = VTMP1;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
usdot(DestTmp.Z(), Vector1.Z(), Vector2.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
usdot(DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSDot) {
|
||||
///< Dest = Acc + dot(Vector1 (signed 8-bit), Vector2 (signed 8-bit))
|
||||
// Matches:
|
||||
// - SVE - SDOT
|
||||
// - ASIMD - SDOT
|
||||
const auto Op = IROp->C<IR::IROp_VSDot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// SDOT accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
DestTmp = Dst;
|
||||
} else {
|
||||
DestTmp = VTMP1;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Z(), Vector1.Z(), Vector2.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSAddLP) {
|
||||
const auto Op = IROp->C<IR::IROp_VSAddLP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// SVE only has the accumulating form, so accumulate in to a zeroed register.
|
||||
// Zero a temporary instead if Dst aliases the source.
|
||||
const auto DestTmp = Dst == Vector ? VTMP1 : Dst;
|
||||
dup_imm(SubRegSize, DestTmp.Z(), 0);
|
||||
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
saddlp(SubRegSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSAdALP) {
|
||||
const auto Op = IROp->C<IR::IROp_VSAdALP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
// SADALP accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
DestTmp = Dst != Vector ? Dst : VTMP1;
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst == Vector) {
|
||||
// ASIMD has the non-accumulating form, which is cheaper than shuffling through a temporary.
|
||||
saddlp(SubRegSize, VTMP1.Q(), Vector.Q());
|
||||
add(SubRegSize, Dst.Q(), Acc.Q(), VTMP1.Q());
|
||||
return;
|
||||
}
|
||||
|
||||
if (Dst != Acc) {
|
||||
mov(Dst.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
sadalp(SubRegSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUMull) {
|
||||
const auto Op = IROp->C<IR::IROp_VUMull>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -4451,13 +4658,9 @@ DEF_OP(VFMLS) {
|
||||
|
||||
if (Is128Bit) {
|
||||
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
}
|
||||
|
||||
if (Is128Bit) {
|
||||
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
fmla(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
|
||||
@@ -4611,13 +4814,9 @@ DEF_OP(VFNMLS) {
|
||||
|
||||
if (Is128Bit) {
|
||||
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
}
|
||||
|
||||
if (Is128Bit) {
|
||||
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
fmls(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
|
||||
@@ -4631,6 +4830,106 @@ DEF_OP(VFNMLS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VBlendImm) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VBlendImm>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto Selector = Op->Selector;
|
||||
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto LHS = GetVReg(Op->LHS);
|
||||
const auto RHS = GetVReg(Op->RHS);
|
||||
const auto DstIsNonAliasing = Dst != LHS && Dst != RHS;
|
||||
|
||||
// Silly case where two blending sources are the same.
|
||||
if (LHS == RHS) {
|
||||
if (DstIsNonAliasing) {
|
||||
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// We'll need to expand our selector to match its predicate equivalent.
|
||||
// The lowest bit of each predicate element being set to 1 signifies
|
||||
// that it's enabled.
|
||||
const auto MakePredicateMask = [ElementSize, Is256Bit, OpSize](uint16_t Imm) {
|
||||
if (ElementSize == IR::OpSize::i8Bit) {
|
||||
// Since we use a u16 selector, we have enough bits for every byte in a
|
||||
// 128-bit lane, so we don't need to do anything here except replicate the
|
||||
// bits in the event of 256-bit.
|
||||
return Is256Bit ? uint32_t(Imm) << 16 | Imm : Imm;
|
||||
}
|
||||
|
||||
uint32_t Mask = 0;
|
||||
const auto DataSize = IR::OpSizeToSize(ElementSize);
|
||||
const auto NumElements = IR::NumElements(OpSize, ElementSize);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
if (((Imm >> i) & 1) != 0) {
|
||||
Mask |= 1U << (DataSize * i);
|
||||
}
|
||||
}
|
||||
return Mask;
|
||||
};
|
||||
|
||||
// Our predicate that we'll be firing our constructed bitmask into.
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0.Merging();
|
||||
|
||||
// TODO: We can completely eliminate this via PMOV in SVE2.1
|
||||
ARMEmitter::ForwardLabel AfterLabel;
|
||||
ARMEmitter::BackwardLabel ConstantLabel;
|
||||
(void)b(&AfterLabel);
|
||||
(void)Bind(&ConstantLabel);
|
||||
const auto PredicateMask = MakePredicateMask(Selector);
|
||||
if (Dst == RHS) {
|
||||
dc32(~PredicateMask);
|
||||
} else {
|
||||
dc32(PredicateMask);
|
||||
}
|
||||
(void)Bind(&AfterLabel);
|
||||
(void)adr(TMP1, &ConstantLabel);
|
||||
ldr(Predicate, TMP1);
|
||||
|
||||
if (Dst == LHS) {
|
||||
mov(SubRegSize, LHS.Z(), Predicate, RHS.Z());
|
||||
} else if (Dst == RHS) {
|
||||
mov(SubRegSize, RHS.Z(), Predicate, LHS.Z());
|
||||
} else {
|
||||
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
|
||||
mov(SubRegSize, Dst.Z(), Predicate, RHS.Z());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VXar) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VXar>();
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto LHS = GetVReg(Op->LHS);
|
||||
const auto RHS = GetVReg(Op->RHS);
|
||||
const auto Rotate = Op->Rotate;
|
||||
LOGMAN_THROW_A_FMT(Rotate >= 1 && Rotate <= ElementSizeBits, "Rotate immediate must be within [1, {}]", ElementSizeBits);
|
||||
|
||||
if (Dst == LHS) {
|
||||
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
|
||||
} else if (Dst == RHS) {
|
||||
movprfx(VTMP1.Z(), LHS.Z());
|
||||
xar(SubRegSize, VTMP1.Z(), RHS.Z(), Rotate);
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
movprfx(Dst.Z(), LHS.Z());
|
||||
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFCopySign) {
|
||||
auto Op = IROp->C<IR::IROp_VFCopySign>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <span>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
@@ -93,13 +94,15 @@ struct GuestToHostMap {
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second;
|
||||
return BlockList
|
||||
.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, fextl::vector<uint64_t>(CodePages.begin(), CodePages.end())})
|
||||
.first->second;
|
||||
}
|
||||
|
||||
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
|
||||
@@ -220,7 +223,7 @@ public:
|
||||
}
|
||||
|
||||
if (HostPtr && DynamicL1Cache()) {
|
||||
UpdateDynamicL1Stats(Thread);
|
||||
UpdateDynamicL1Stats(Thread, Address, HostPtr);
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
|
||||
@@ -228,7 +231,7 @@ public:
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddress, uint64_t HostCode) {
|
||||
// If host pointer was found in L2 or L3, then add it to the counter.
|
||||
// Keeping track not L1 misses, but specifically L2/L3 hits.
|
||||
++L2L3CacheHits;
|
||||
@@ -242,12 +245,18 @@ public:
|
||||
|
||||
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries < MAX_L1_ENTRIES) {
|
||||
// Entries whose address has the new mask bit set would be unreachable by InvalidateCache
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(L1Pointer), CurrentL1Entries * sizeof(LookupCacheEntry), false);
|
||||
|
||||
CurrentL1Entries <<= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
|
||||
// If L1 was just shrunk, then we just removed our cached entry. Add it back.
|
||||
AddL1Entry(GuestAddress, HostCode);
|
||||
}
|
||||
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries > MIN_L1_ENTRIES) {
|
||||
@@ -275,7 +284,7 @@ public:
|
||||
|
||||
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, auto& Addresses, uint64_t Start, uint64_t Length) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
@@ -285,7 +294,7 @@ public:
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode) {
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
@@ -380,15 +389,19 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
void AddL1Entry(uint64_t GuestAddress, uint64_t HostCode) {
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[GuestAddress & L1PointerMask];
|
||||
L1Entry.GuestCode = GuestAddress;
|
||||
L1Entry.HostCode = HostCode;
|
||||
}
|
||||
|
||||
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
|
||||
for (const auto& CodePage : Entry.CodePages) {
|
||||
CachedCodePages[CodePage >> 12].insert(Address);
|
||||
}
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = Entry.HostCode;
|
||||
AddL1Entry(Address, Entry.HostCode);
|
||||
|
||||
if (!DisableL2Cache() && !L1Only) {
|
||||
// Do ful map
|
||||
|
||||
@@ -36,35 +36,6 @@ using X86Tables::OpToIndex;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
constexpr size_t SyscallArgs = 7;
|
||||
using SyscallArray = std::array<uint64_t, SyscallArgs>;
|
||||
|
||||
size_t NumArguments {};
|
||||
const SyscallArray* GPRIndexes {};
|
||||
static constexpr SyscallArray GPRIndexes_64 = {
|
||||
FEXCore::X86State::REG_RAX, FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_R10, FEXCore::X86State::REG_R8, FEXCore::X86State::REG_R9,
|
||||
};
|
||||
static constexpr SyscallArray GPRIndexes_32 = {
|
||||
FEXCore::X86State::REG_RAX, FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RCX, FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_RBP,
|
||||
};
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64) {
|
||||
NumArguments = GPRIndexes_64.size();
|
||||
GPRIndexes = &GPRIndexes_64;
|
||||
} else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX32) {
|
||||
NumArguments = GPRIndexes_32.size();
|
||||
GPRIndexes = &GPRIndexes_32;
|
||||
} else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// All registers will be spilled before the syscall and filled afterwards so no JIT-side argument handling is necessary.
|
||||
NumArguments = 0;
|
||||
GPRIndexes = nullptr;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unhandled OSABI syscall");
|
||||
}
|
||||
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
@@ -72,13 +43,6 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
auto NewRIP = GetRelocatedPC(Op, -Op->InstSize);
|
||||
_StoreContextGPR(GPRSize, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
|
||||
Ref Arguments[SyscallArgs] {
|
||||
InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode,
|
||||
};
|
||||
for (size_t i = 0; i < NumArguments; ++i) {
|
||||
Arguments[i] = LoadGPRRegister(GPRIndexes->at(i));
|
||||
}
|
||||
|
||||
if (IsSyscallInst) {
|
||||
// If this is the `Syscall` instruction rather than `int 0x80` then we need to do some additional work.
|
||||
// RCX = RIP after this instruction
|
||||
@@ -94,18 +58,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
}
|
||||
|
||||
FlushRegisterCache();
|
||||
auto SyscallOp = _Syscall(Arguments[0], Arguments[1], Arguments[2], Arguments[3], Arguments[4], Arguments[5], Arguments[6]);
|
||||
|
||||
// Generic ABI doesn't store result in RAX.
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
StoreGPRRegister(X86State::REG_RAX, SyscallOp);
|
||||
}
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, rip));
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
_Syscall();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
@@ -268,10 +221,10 @@ void OpDispatchBuilder::SecondaryALUOp(OpcodeArgs) {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
ALUOp(Op, IROp, AtomicIROp, 1);
|
||||
ALUOp(Op, IROp, AtomicIROp, 1, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
@@ -286,6 +239,8 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Before = _AtomicFetchAdd(Size, ALUOp, DestMem);
|
||||
} else if (DestRAX) {
|
||||
Before = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
@@ -302,12 +257,14 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Result = CalculateFlags_ADC(Size, Before, Src);
|
||||
}
|
||||
|
||||
if (!DestIsLockedMem(Op)) {
|
||||
if (DestRAX) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Result, OpSizeFromDst(Op));
|
||||
} else if (!DestIsLockedMem(Op)) {
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
@@ -323,13 +280,17 @@ void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
auto SrcPlusCF = IncrementByCarry(OpSize, Src);
|
||||
Before = _AtomicFetchSub(Size, SrcPlusCF, DestMem);
|
||||
} else if (DestRAX) {
|
||||
Before = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
|
||||
Result = CalculateFlags_SBB(Size, Before, Src);
|
||||
|
||||
if (!DestIsLockedMem(Op)) {
|
||||
if (DestRAX) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Result, OpSizeFromDst(Op));
|
||||
} else if (!DestIsLockedMem(Op)) {
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
}
|
||||
@@ -339,7 +300,8 @@ void OpDispatchBuilder::SALCOp(OpcodeArgs) {
|
||||
|
||||
auto Result = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, _InlineConstant(0xffffffff), _InlineConstant(0));
|
||||
|
||||
StoreResultGPR(Op, Result);
|
||||
// This inserts in to the low 8-bits.
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, OpSizeFromDst(Op));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
|
||||
@@ -814,11 +776,11 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
OpSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[1].Literal();
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[0].Literal();
|
||||
|
||||
Ref CondReg = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Ref CondReg = LoadGPRRegister(X86State::REG_RCX, SrcSize);
|
||||
CondReg = Sub(OpSize, CondReg, 1);
|
||||
StoreResultGPR(Op, Op->Src[0], CondReg);
|
||||
StoreGPRRegister(X86State::REG_RCX, CondReg, SrcSize);
|
||||
|
||||
// If LOOPE then jumps to target if RCX != 0 && ZF == 1
|
||||
// If LOOPNE then jumps to target if RCX != 0 && ZF == 0
|
||||
@@ -847,7 +809,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
StartNewBlock();
|
||||
|
||||
// Store the new RIP
|
||||
ExitRelocatedPC(Op, Op->Src[1].Literal());
|
||||
ExitRelocatedPC(Op, Op->Src[0].Literal());
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -1002,11 +964,16 @@ void OpDispatchBuilder::RETFARIndirectOp(OpcodeArgs) {
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// TEST is an instruction that does an AND between the sources
|
||||
// Result isn't stored in result, only writes to flags
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest {};
|
||||
if (DestRAX) {
|
||||
Dest = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
@@ -1112,26 +1079,60 @@ void OpDispatchBuilder::MOVZXOp(OpcodeArgs) {
|
||||
StoreResultGPR(Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// CMP is an instruction that does a SUB between the sources
|
||||
// Result isn't stored in result, only writes to flags
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest {};
|
||||
if (DestRAX) {
|
||||
Dest = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Size = OpSizeFromSrc(Op);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
Ref Upper = _Sbfe(std::max(OpSize::i32Bit, Size), 1, GetSrcBitSize(Op) - 1, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RDX, Upper, Size);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Upper);
|
||||
std::optional<Ref> OpDispatchBuilder::XCHGOpImpl(OpcodeArgs, Ref Src) {
|
||||
if (DestIsMem(Op)) {
|
||||
HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
if (IsMonoBackpatcherBlock) {
|
||||
_MonoBackpatcherWrite(OpSizeFromSrc(Op), Src, Dest);
|
||||
} else {
|
||||
return _AtomicSwap(OpSizeFromSrc(Op), Src, Dest);
|
||||
}
|
||||
return std::nullopt;
|
||||
} else {
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Swap the contents
|
||||
// Order matters here since we don't want to swap context contents for one that effects the other
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
return Dest;
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Res = XCHGOpImpl(Op, Src);
|
||||
if (Res) {
|
||||
StoreResultGPR(Op, Op->Src[0], *Res);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XCHGRAXOp(OpcodeArgs) {
|
||||
// Load both the source and the destination
|
||||
if (Op->OP == 0x90 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX && Op->Dest.IsGPR() &&
|
||||
Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
if (Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// This is one heck of a sucky special case
|
||||
// If we are the 0x90 XCHG opcode (Meaning source is GPR RAX)
|
||||
// and destination register is ALSO RAX
|
||||
@@ -1159,25 +1160,10 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
if (DestIsMem(Op)) {
|
||||
HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
if (IsMonoBackpatcherBlock) {
|
||||
_MonoBackpatcherWrite(OpSizeFromSrc(Op), Src, Dest);
|
||||
} else {
|
||||
auto Result = _AtomicSwap(OpSizeFromSrc(Op), Src, Dest);
|
||||
StoreResultGPR(Op, Op->Src[0], Result);
|
||||
}
|
||||
} else {
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Swap the contents
|
||||
// Order matters here since we don't want to swap context contents for one that effects the other
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
StoreResultGPR(Op, Op->Src[0], Dest);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, OpSize::iInvalid, 0, true);
|
||||
auto Res = XCHGOpImpl(Op, Src);
|
||||
if (Res) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, *Res, OpSizeFromDst(Op));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1188,7 +1174,8 @@ void OpDispatchBuilder::CDQOp(OpcodeArgs) {
|
||||
|
||||
Src = _Sbfe(DstSize <= OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, IR::OpSizeAsBits(SrcSize), 0, Src);
|
||||
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize);
|
||||
// This inserts in to the low 16-bits.
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, DstSize);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
@@ -1359,27 +1346,25 @@ void OpDispatchBuilder::MOVOffsetOp(OpcodeArgs) {
|
||||
// Source is memory(literal)
|
||||
// Dest is GPR
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.ForceLoad = true});
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, OpSizeFromDst(Op));
|
||||
break;
|
||||
}
|
||||
case 0xA2:
|
||||
case 0xA3: {
|
||||
// Source is GPR
|
||||
// Dest is memory(literal)
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX);
|
||||
|
||||
// This one is a bit special since the destination is a literal
|
||||
// So the destination gets stored in Src[1]
|
||||
StoreResultGPR(Op, Op->Src[1], Src);
|
||||
StoreResultGPR(Op, Op->Src[0], Src);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
Ref Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], GPRSize, Op->Flags);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Leaf = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
Ref RAX = _AllocateGPR(false);
|
||||
@@ -1421,7 +1406,7 @@ void OpDispatchBuilder::XGetBVOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHLOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
Ref Result = _Lshl(Size == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Dest, Src);
|
||||
HandleShift(Op, Result, Dest, ShiftType::LSL, Src);
|
||||
@@ -1443,7 +1428,7 @@ void OpDispatchBuilder::SHLImmediateOp(OpcodeArgs, bool SHL1Bit) {
|
||||
void OpDispatchBuilder::SHROp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
auto ALUOp = _Lshr(std::max(OpSize::i32Bit, Size), Dest, Src);
|
||||
HandleShift(Op, ALUOp, Dest, ShiftType::LSR, Src);
|
||||
@@ -1472,7 +1457,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
// Allow garbage on the shift, we're masking it anyway.
|
||||
Ref Shift = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Shift = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op.
|
||||
if (Size == 64) {
|
||||
@@ -1526,7 +1511,7 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
Res = _Extr(OpSizeFromSrc(Op), Dest, Src, Size - Shift);
|
||||
}
|
||||
|
||||
CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Res, Dest, Shift);
|
||||
CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Res, Dest, Shift, true);
|
||||
CalculateDeferredFlags();
|
||||
StoreResultGPR(Op, Res);
|
||||
} else if (Shift == 0 && Size == 32) {
|
||||
@@ -1543,7 +1528,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
Ref Shift = LoadGPRRegister(X86State::REG_RCX);
|
||||
Ref Shift = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
const auto Size = GetDstBitSize(Op);
|
||||
|
||||
@@ -1621,7 +1606,7 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs, bool Immediate, bool SHR1Bit) {
|
||||
CalculateDeferredFlags();
|
||||
StoreResultGPR(Op, Result);
|
||||
} else {
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref Result = _Ashr(OpSize, Dest, Src);
|
||||
|
||||
HandleShift(Op, Result, Dest, ShiftType::ASR, Src);
|
||||
@@ -1645,7 +1630,7 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
UnmaskedConst = GetConstantShift(Op, Is1Bit);
|
||||
UnmaskedSrc = ARef(UnmaskedConst);
|
||||
} else {
|
||||
UnmaskedSrc = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
UnmaskedSrc = ARef(LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true));
|
||||
}
|
||||
auto Src = UnmaskedSrc.And(Mask);
|
||||
|
||||
@@ -1689,8 +1674,8 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ANDNBMIOp(OpcodeArgs) {
|
||||
auto* Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto* Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
auto Dest = _Andn(OpSizeFromSrc(Op), Src2, Src1);
|
||||
|
||||
@@ -1703,8 +1688,8 @@ void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) {
|
||||
// along with some edge-case handling and flag setting.
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
|
||||
auto* Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto* Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto SrcSize = IR::OpSizeAsBits(Size);
|
||||
@@ -1746,7 +1731,7 @@ void OpDispatchBuilder::BLSIBMIOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto NegatedSrc = _Neg(Size, Src);
|
||||
auto Result = _And(Size, Src, NegatedSrc);
|
||||
|
||||
@@ -1767,14 +1752,14 @@ void OpDispatchBuilder::BLSMSKBMIOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = _Xor(Size, Sub(Size, Src, 1), Src);
|
||||
|
||||
StoreResultGPR(Op, Result);
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF set according to the Src
|
||||
auto CFInv = To01(OpSize::i64Bit, Src);
|
||||
auto CFInv = To01(Size, Src);
|
||||
|
||||
// The output of BLSMSK is always nonzero, so TST will clear Z (along with C
|
||||
// and O) while setting S.
|
||||
@@ -1787,12 +1772,12 @@ void OpDispatchBuilder::BLSRBMIOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = _And(Size, Sub(Size, Src, 1), Src);
|
||||
|
||||
StoreResultGPR(Op, Result);
|
||||
|
||||
auto CFInv = To01(OpSize::i64Bit, Src);
|
||||
auto CFInv = To01(Size, Src);
|
||||
|
||||
SetNZ_ZeroCV(Size, Result);
|
||||
SetCFInverted(CFInv);
|
||||
@@ -1807,8 +1792,8 @@ void OpDispatchBuilder::BMI2Shift(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? GPRSize : Size;
|
||||
|
||||
auto* Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
auto* Shift = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GPRSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
auto Shift = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GPRSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Ref Result;
|
||||
if (Op->OP == 0x6F7) {
|
||||
@@ -1831,9 +1816,9 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
|
||||
|
||||
// In 32-bit mode we only look at bottom 32-bit, no 8 or 16-bit BZHI so no
|
||||
// need to zero-extend sources
|
||||
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
auto* Index = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Index = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Clear the high bits specified by the index. A64 only considers bottom bits
|
||||
// of the shift, so we don't need to mask bottom 8-bits ourselves.
|
||||
@@ -1878,8 +1863,8 @@ void OpDispatchBuilder::RORX(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto* Result = Src;
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = Src;
|
||||
if (DoRotation) [[likely]] {
|
||||
Result = _Ror(OpSizeFromSrc(Op), Src, _InlineConstant(Amount));
|
||||
}
|
||||
@@ -1916,8 +1901,8 @@ void OpDispatchBuilder::MULX(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::PDEP(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
|
||||
auto* Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto* Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = _PDep(OpSizeFromSrc(Op), Input, Mask);
|
||||
|
||||
StoreResultGPR(Op, Op->Dest, Result);
|
||||
@@ -1925,8 +1910,8 @@ void OpDispatchBuilder::PDEP(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::PEXT(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
|
||||
auto* Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto* Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = _PExt(OpSizeFromSrc(Op), Input, Mask);
|
||||
|
||||
StoreResultGPR(Op, Op->Dest, Result);
|
||||
@@ -1936,8 +1921,8 @@ void OpDispatchBuilder::ADXOp(OpcodeArgs) {
|
||||
const auto OpSize = OpSizeFromSrc(Op);
|
||||
|
||||
// Only 32/64-bit anyway so allow garbage, we use 32-bit ops.
|
||||
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto* Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Handles ADCX and ADOX
|
||||
const bool IsADCX = Op->OP == 0x1F6;
|
||||
@@ -2035,11 +2020,11 @@ void OpDispatchBuilder::RCROp8x1Bit(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(_XorShift(OpSize::i32Bit, Res, Res, ShiftType::LSR, 1), SizeBit - 2, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCROp(OpcodeArgs, bool UseRCX) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
if (Size == 8 || Size == 16) {
|
||||
RCRSmallerOp(Op);
|
||||
RCRSmallerOp(Op, UseRCX);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2049,49 +2034,54 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
const auto OpSize = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
if (!UseRCX) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Res = Src >> Shift
|
||||
Ref Res = _Lshr(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Constant folded version of the above, with fused shifts.
|
||||
if (Const > 1) {
|
||||
Res = _Orlshl(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source.
|
||||
SetCFDirect(Dest, Const - 1, true);
|
||||
|
||||
// Since shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Size - Const);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto Xor = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, Size - 2, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Res = Src >> Shift
|
||||
Ref Res = _Lshr(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Constant folded version of the above, with fused shifts.
|
||||
if (Const > 1) {
|
||||
Res = _Orlshl(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source.
|
||||
SetCFDirect(Dest, Const - 1, true);
|
||||
|
||||
// Since shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Size - Const);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto Xor = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, Size - 2, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref SrcMasked = _And(OpSize, Src, _InlineConstant(Mask));
|
||||
|
||||
Calculate_ShiftVariable(
|
||||
Op, SrcMasked,
|
||||
[this, Op, Size, OpSize]() {
|
||||
// Rematerialize loads to avoid crossblock liveness
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Res = Src >> Shift
|
||||
@@ -2127,21 +2117,29 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
OpSizeFromSrc(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs, bool UseRCX) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto GetShift = [this, Op, UseRCX]() {
|
||||
if (UseRCX) {
|
||||
auto Src = ARef(LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true));
|
||||
return Src.And(0x1F);
|
||||
} else {
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
return Src.And(0x1F);
|
||||
}
|
||||
};
|
||||
|
||||
auto Src = GetShift();
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size, GetShift]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto Src = GetShift();
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -2251,11 +2249,11 @@ void OpDispatchBuilder::RCLOp1Bit(OpcodeArgs) {
|
||||
StoreResultGPR(Op, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCLOp(OpcodeArgs, bool UseRCX) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
if (Size == 8 || Size == 16) {
|
||||
RCLSmallerOp(Op);
|
||||
RCLSmallerOp(Op, UseRCX);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2264,50 +2262,55 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
const auto OpSize = OpSizeFromSrc(Op);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
if (!UseRCX) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
// Res = Src << Shift
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Res = _Lshl(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
if (Const > 1) {
|
||||
Res = _Orlshr(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
SetCFDirect(Dest, Size - Const, true);
|
||||
|
||||
// Since Shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Const - 1);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto NewOF = _Xor(OpSize, Res, Dest);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Size - 1, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
// Res = Src << Shift
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Res = _Lshl(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
if (Const > 1) {
|
||||
Res = _Orlshr(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
SetCFDirect(Dest, Size - Const, true);
|
||||
|
||||
// Since Shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Const - 1);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto NewOF = _Xor(OpSize, Res, Dest);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Size - 1, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref SrcMasked = _And(OpSize, Src, _InlineConstant(Mask));
|
||||
|
||||
Calculate_ShiftVariable(
|
||||
Op, SrcMasked,
|
||||
[this, Op, Size, OpSize]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
// Res = Src << Shift
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
@@ -2342,21 +2345,29 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
OpSizeFromSrc(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs, bool UseRCX) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto GetShift = [this, Op, UseRCX]() {
|
||||
if (UseRCX) {
|
||||
auto Src = ARef(LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true));
|
||||
return Src.And(0x1F);
|
||||
} else {
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
return Src.And(0x1F);
|
||||
}
|
||||
};
|
||||
|
||||
auto Src = GetShift();
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size, GetShift]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto Src = GetShift();
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
@@ -3214,7 +3225,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
|
||||
if (!Repeat) {
|
||||
// Src is used only for a store of the same size so allow garbage
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
|
||||
// Only ES prefix
|
||||
Ref Dest = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
@@ -3233,7 +3244,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
// FEX doesn't support partial faulting REP instructions.
|
||||
// Converting this to a `MemSet` IR op optimizes this quite significantly in our codegen.
|
||||
// If FEX is to gain support for faulting REP instructions, then this implementation needs to change significantly.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Dest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
// Only ES prefix
|
||||
@@ -3470,7 +3481,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
StoreResultGPR(Op, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, Size);
|
||||
|
||||
// Offset the pointer
|
||||
Ref TailDest_RSI = OffsetByDir(Src_RSI, IR::OpSizeToSize(Size));
|
||||
@@ -3484,7 +3495,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size, AddrSize](int32_t PtrDir) {
|
||||
ForeachDirection([this, Size, AddrSize](int32_t PtrDir) {
|
||||
// XXX: Theoretically LODS could be optimized to
|
||||
// RSI += {-}(RCX * Size)
|
||||
// RAX = [RSI - Size]
|
||||
@@ -3517,7 +3528,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
StoreResultGPR(Op, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, Size);
|
||||
|
||||
Ref TailCounter = LoadGPRRegister(X86State::REG_RCX);
|
||||
Ref TailDest_RSI = LoadGPRRegister(X86State::REG_RSI);
|
||||
@@ -3564,7 +3575,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src1 = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2);
|
||||
@@ -3607,7 +3618,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src1 = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2);
|
||||
@@ -3877,7 +3888,6 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
// This allows us to only hit the ZEXT case on failure
|
||||
Ref RAXResult = NZCVSelect(OpSize::i64Bit, CondClass::EQ, Src3, Src1Lower);
|
||||
|
||||
// When the size is 4 we need to make sure not zext the GPR when the comparison fails
|
||||
StoreGPRRegister(X86State::REG_RAX, RAXResult);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RAX, Src1Lower, Size);
|
||||
@@ -3891,7 +3901,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src2Lower = _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src2);
|
||||
}
|
||||
Ref DestResult = Trivial ? Src2 : NZCVSelect(OpSize::i64Bit, CondClass::EQ, Src2Lower, Src1);
|
||||
Ref DestResult = Trivial ? Src2Lower : NZCVSelect(OpSize::i64Bit, CondClass::EQ, Src2Lower, Src1);
|
||||
|
||||
// Store in to GPR Dest
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
@@ -4350,12 +4360,23 @@ AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, con
|
||||
if (Operand.IsGPRIndirectRelocation()) {
|
||||
A.Base = Add(GPRSize, _EntrypointOffset(GPRSize, Operand.Data.GPRIndirect.Displacement), A.Base);
|
||||
} else {
|
||||
A.Offset = static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement);
|
||||
if (Operand.Data.GPRIndirect.PatchableDisp) {
|
||||
A.Base = Add(GPRSize, A.Base,
|
||||
_PatchableGuestData(OpSize::i64Bit, static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement),
|
||||
Op->PC + Operand.Data.GPRIndirect.DispOffset, 4));
|
||||
} else {
|
||||
A.Offset = static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement);
|
||||
}
|
||||
}
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.GPRIndirect.GPR);
|
||||
} else if (Operand.IsRIPRelative() || Operand.IsRIPRelativeRelocation()) {
|
||||
if (Is64BitMode) {
|
||||
A.Base = GetRelocatedPC(Op, static_cast<int32_t>(Operand.Data.RIPLiteral.Value));
|
||||
if (Operand.IsRIPRelative() && Operand.Data.RIPLiteral.PatchableDisp) {
|
||||
A.Base = _PatchableGuestRIP(OpSize::i64Bit, Op->PC + Op->InstSize + static_cast<int32_t>(Operand.Data.RIPLiteral.Value),
|
||||
Op->PC + Operand.Data.RIPLiteral.DispOffset, 4);
|
||||
} else {
|
||||
A.Base = GetRelocatedPC(Op, static_cast<int32_t>(Operand.Data.RIPLiteral.Value));
|
||||
}
|
||||
} else {
|
||||
// 32bit this isn't RIP relative but instead absolute
|
||||
if (Operand.IsRIPRelativeRelocation()) {
|
||||
@@ -4392,6 +4413,14 @@ AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, con
|
||||
} else {
|
||||
A.Base = EPOffset;
|
||||
}
|
||||
} else if (Operand.Data.SIB.PatchableDisp) {
|
||||
Ref PatchedDisp =
|
||||
_PatchableGuestData(OpSize::i64Bit, static_cast<int32_t>(Operand.Data.SIB.Offset), Op->PC + Operand.Data.SIB.DispOffset, 4);
|
||||
if (A.Base) {
|
||||
A.Base = Add(OpSize::i64Bit, A.Base, PatchedDisp);
|
||||
} else {
|
||||
A.Base = PatchedDisp;
|
||||
}
|
||||
} else {
|
||||
A.Offset = static_cast<int32_t>(Operand.Data.SIB.Offset);
|
||||
}
|
||||
@@ -4399,6 +4428,9 @@ AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, con
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.SIB.Base) || IsNonTSOReg(AccessType, Operand.Data.SIB.Index);
|
||||
} else if (Operand.IsLiteralRelocation()) {
|
||||
A.Base = _EntrypointOffset(GPRSize, Operand.Data.LiteralRelocation.EntrypointOffset);
|
||||
} else if (Operand.IsLiteralPatchable()) {
|
||||
A.Base = _PatchableGuestData(OpSize::i64Bit, Operand.Data.LiteralPatchable.Value, Op->PC + Operand.Data.LiteralPatchable.FieldOffset,
|
||||
static_cast<uint64_t>(Operand.Data.LiteralPatchable.Width));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unknown Src Type: {}\n", Operand.Type);
|
||||
}
|
||||
@@ -4618,7 +4650,7 @@ void OpDispatchBuilder::StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref
|
||||
}
|
||||
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9}
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9 != 0}
|
||||
, CTX {ctx}
|
||||
, Thread {Thread} {
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
@@ -4660,7 +4692,7 @@ void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
|
||||
StoreResultGPR(Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx) {
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx, bool DestRAX) {
|
||||
// On x86, the canonical way to zero a register is XOR with itself. Detect and
|
||||
// emit optimal arm64 assembly.
|
||||
if (!DestIsLockedMem(Op) && ALUIROp == FEXCore::IR::IROps::OP_XOR && Op->Dest.IsGPR() && Op->Src[SrcIdx].IsGPR() &&
|
||||
@@ -4696,7 +4728,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
// promoting to a full size operation that preserves the upper bits.
|
||||
uint64_t Const;
|
||||
bool IsConst = IsValueConstant(WrapNode(Src), &Const);
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits && IsConst &&
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && ((Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits) || DestRAX) && IsConst &&
|
||||
(ALUIROp == IR::IROps::OP_XOR || ALUIROp == IR::IROps::OP_OR || ALUIROp == IR::IROps::OP_ANDWITHFLAGS)) {
|
||||
|
||||
RoundedSize = ResultSize = GetGPROpSize();
|
||||
@@ -4725,6 +4757,8 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
DeriveOp(FetchOp, AtomicFetchOp, _AtomicFetchAdd(Size, Src, DestMem));
|
||||
Dest = FetchOp;
|
||||
} else if (DestRAX) {
|
||||
Dest = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
@@ -4759,7 +4793,9 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
default: break;
|
||||
}
|
||||
|
||||
if (!DestIsLockedMem(Op)) {
|
||||
if (DestRAX) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Result, ResultSize);
|
||||
} else if (!DestIsLockedMem(Op)) {
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, ResultSize, OpSize::iInvalid, MemoryAccessType::DEFAULT);
|
||||
}
|
||||
}
|
||||
@@ -5016,8 +5052,7 @@ void OpDispatchBuilder::CLZeroOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref DestMem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
_CacheLineZero(DestMem);
|
||||
_CacheLineZero(LoadGPRRegister(X86State::REG_RAX));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::Prefetch(OpcodeArgs, bool ForStore, bool Stream, uint8_t Level) {
|
||||
@@ -5038,7 +5073,11 @@ void OpDispatchBuilder::RDTSCPOp(OpcodeArgs) {
|
||||
// - Explicitly use an MFENCE before this instruction if you want this behaviour
|
||||
// This instruction is not an execution fence, so subsequent instructions can execute after this
|
||||
// - Explicitly use an LFENCE after RDTSCP if you want to block this behaviour
|
||||
|
||||
if (CTX->HostFeatures.HostType != FEXCore::HostFeatures::HostTypeEnum::Linux && !CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
// RDTSCP is unsupported on Win32 platforms if TPIDRRO isn't supported.
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
auto Counter = CycleCounter(true);
|
||||
|
||||
auto ID = _ProcessorID();
|
||||
@@ -5048,6 +5087,11 @@ void OpDispatchBuilder::RDTSCPOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RDPIDOp(OpcodeArgs) {
|
||||
if (CTX->HostFeatures.HostType != FEXCore::HostFeatures::HostTypeEnum::Linux && !CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
// RDTSCP is unsupported on Win32 platforms if TPIDRRO isn't supported.
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
StoreResultGPR(Op, _ProcessorID());
|
||||
}
|
||||
|
||||
@@ -5075,7 +5119,7 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RDRANDOp(OpcodeArgs, bool Reseed) {
|
||||
if (!CTX->HostFeatures.SupportsRAND) {
|
||||
if (!CTX->HostFeatures.SupportsRAND && !CTX->SoftwareRNGEnabled()) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -202,9 +202,10 @@ public:
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, InvalidNode, InvalidNode);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock,
|
||||
uint64_t PatchSiteAddress = 0, uint64_t PatchSiteSize = 0) {
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock);
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock, PatchSiteAddress, PatchSiteSize);
|
||||
}
|
||||
IRPair<IROp_Break> Break(BreakDefinition Reason) {
|
||||
FlushRegisterCache();
|
||||
@@ -360,8 +361,10 @@ public:
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedNoNopOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs, bool IsAVX);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void ALURAXOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx, bool DestRAX);
|
||||
void LSLOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
void SyscallOp(OpcodeArgs, bool IsSyscallInst);
|
||||
@@ -372,8 +375,8 @@ public:
|
||||
void IRETOp(OpcodeArgs);
|
||||
void CallbackReturnOp(OpcodeArgs);
|
||||
void SecondaryALUOp(OpcodeArgs);
|
||||
void ADCOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void SBBOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void ADCOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void SBBOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void SALCOp(OpcodeArgs);
|
||||
void PUSHOp(OpcodeArgs);
|
||||
void PUSHREGOp(OpcodeArgs);
|
||||
@@ -393,16 +396,18 @@ public:
|
||||
void JUMPFARIndirectOp(OpcodeArgs);
|
||||
void CALLFARIndirectOp(OpcodeArgs);
|
||||
void RETFARIndirectOp(OpcodeArgs);
|
||||
void TESTOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void TESTOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void ARPLOp(OpcodeArgs);
|
||||
void MOVSXDOp(OpcodeArgs);
|
||||
void MOVSXOp(OpcodeArgs);
|
||||
void MOVZXOp(OpcodeArgs);
|
||||
void CMPOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void CMPOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void SETccOp(OpcodeArgs);
|
||||
void CQOOp(OpcodeArgs);
|
||||
void CDQOp(OpcodeArgs);
|
||||
std::optional<Ref> XCHGOpImpl(OpcodeArgs, Ref Src);
|
||||
void XCHGOp(OpcodeArgs);
|
||||
void XCHGRAXOp(OpcodeArgs);
|
||||
void SAHFOp(OpcodeArgs);
|
||||
void LAHFOp(OpcodeArgs);
|
||||
void MOVSegOp(OpcodeArgs, bool ToSeg);
|
||||
@@ -424,11 +429,11 @@ public:
|
||||
void RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool Is1Bit);
|
||||
void RCROp1Bit(OpcodeArgs);
|
||||
void RCROp8x1Bit(OpcodeArgs);
|
||||
void RCROp(OpcodeArgs);
|
||||
void RCRSmallerOp(OpcodeArgs);
|
||||
void RCROp(OpcodeArgs, bool UseRCX);
|
||||
void RCRSmallerOp(OpcodeArgs, bool UseRCX);
|
||||
void RCLOp1Bit(OpcodeArgs);
|
||||
void RCLOp(OpcodeArgs);
|
||||
void RCLSmallerOp(OpcodeArgs);
|
||||
void RCLOp(OpcodeArgs, bool UseRCX);
|
||||
void RCLSmallerOp(OpcodeArgs, bool UseRCX);
|
||||
|
||||
void BTOp(OpcodeArgs, uint32_t SrcIndex, enum BTAction Action);
|
||||
|
||||
@@ -663,6 +668,8 @@ public:
|
||||
|
||||
void VPMADDUBSWOp(OpcodeArgs);
|
||||
void VPMADDWDOp(OpcodeArgs);
|
||||
void VPDPBUSDOp(OpcodeArgs, bool Saturating);
|
||||
void VPDPWSSDOp(OpcodeArgs, bool Saturating);
|
||||
|
||||
void VPMASKMOVOp(OpcodeArgs, bool IsStore);
|
||||
|
||||
@@ -739,7 +746,7 @@ public:
|
||||
void X87FLDCW(OpcodeArgs);
|
||||
void X87FNSAVE(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs, bool DestRAX);
|
||||
void X87FRSTOR(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87FXAM(OpcodeArgs);
|
||||
@@ -799,6 +806,7 @@ public:
|
||||
void VPFCMPOp(OpcodeArgs, uint8_t CompType);
|
||||
void PI2FWOp(OpcodeArgs);
|
||||
void PF2IWOp(OpcodeArgs);
|
||||
void PF2IDOp(OpcodeArgs);
|
||||
|
||||
void PMULHRWOp(OpcodeArgs);
|
||||
|
||||
@@ -839,12 +847,12 @@ public:
|
||||
void SHA256MSG2Op(OpcodeArgs);
|
||||
void SHA256RNDS2Op(OpcodeArgs);
|
||||
|
||||
void AESImcOp(OpcodeArgs);
|
||||
void AESImcOp(OpcodeArgs, bool IsAVX);
|
||||
void AESEncOp(OpcodeArgs);
|
||||
void AESEncLastOp(OpcodeArgs);
|
||||
void AESDecOp(OpcodeArgs);
|
||||
void AESDecLastOp(OpcodeArgs);
|
||||
void AESKeyGenAssist(OpcodeArgs);
|
||||
void AESKeyGenAssist(OpcodeArgs, bool IsAVX);
|
||||
|
||||
void VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
void VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
@@ -1033,6 +1041,9 @@ public:
|
||||
|
||||
void AVX128_VPMADDUBSW(OpcodeArgs);
|
||||
void AVX128_VPMADDWD(OpcodeArgs);
|
||||
void AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper);
|
||||
void AVX128_VPDPBUSD(OpcodeArgs, bool Saturating);
|
||||
void AVX128_VPDPWSSD(OpcodeArgs, bool Saturating);
|
||||
|
||||
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
@@ -1381,6 +1392,11 @@ private:
|
||||
const X86Tables::DecodedOperand& Imm, bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask, bool IsAVX);
|
||||
Ref PCMPXSTRXSaturateExplicitLength(IR::OpSize ElementSize, IR::OpSize LengthSize, Ref RawLength);
|
||||
Ref PCMPXSTRXEqualAny(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements);
|
||||
Ref PCMPXSTRXRanges(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements, bool IsSigned);
|
||||
Ref PCMPXSTRXEqualEach(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements);
|
||||
Ref PCMPXSTRXEqualOrdered(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1Length, Ref Src2Length, Ref Src1ValidElements, Ref Indices);
|
||||
|
||||
Ref PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1397,6 +1413,10 @@ private:
|
||||
|
||||
Ref PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
|
||||
|
||||
Ref VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
|
||||
|
||||
Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
|
||||
@@ -1498,6 +1518,20 @@ private:
|
||||
void StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize Size = OpSize::iInvalid, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, const Ref Src);
|
||||
|
||||
// Matches semantics around GPR storing that matches `StoreResult_WithOpSize` behaviour.
|
||||
void StoreGPRResultWithZExtSemantics(uint32_t GPR, Ref Src, IR::OpSize OpSize) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
if (GPRSize == OpSize::i64Bit && OpSize == OpSize::i32Bit) {
|
||||
// If the Source IR op is 64 bits, we need to zext the upper bits
|
||||
// For all other sizes, the upper bits are guaranteed to already be zero
|
||||
Src = GetOpSize(Src) == OpSize::i64Bit ? ARef(Src).Bfe(0, 32).Ref() : Src;
|
||||
StoreGPRRegister(GPR, Src, GPRSize);
|
||||
} else {
|
||||
StoreGPRRegister(GPR, Src, std::min(GPRSize, OpSize));
|
||||
}
|
||||
}
|
||||
|
||||
Ref _GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, bool Inline) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto Offs = Op->PC + Op->InstSize + Offset - Entry;
|
||||
@@ -1508,18 +1542,24 @@ private:
|
||||
return _GetRelocatedPC(Op, Offset, false);
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
uint64_t PatchOffset = 0;
|
||||
uint64_t PatchSize = 0;
|
||||
if (Op->Src[0].IsLiteralPatchable() && Offset && Offset == (int64_t)Op->Src[0].Literal()) {
|
||||
PatchOffset = Op->PC + Op->Src[0].Data.LiteralPatchable.FieldOffset;
|
||||
PatchSize = Op->Src[0].Data.LiteralPatchable.Width;
|
||||
}
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock, PatchOffset, PatchSize);
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
ExitRelocatedPC(Op, Offset, BranchHint::None, InvalidNode, InvalidNode);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
// Literals are immediates as sources but memory addresses as destinations.
|
||||
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation())) && !Operand.IsGPR();
|
||||
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation() || Operand.IsLiteralPatchable())) && !Operand.IsGPR();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -2131,16 +2171,16 @@ private:
|
||||
}
|
||||
|
||||
// Compares two floats and sets flags for a COMISS instruction
|
||||
void Comiss(IR::OpSize ElementSize, Ref Src1, Ref Src2, bool InvalidateAF = false) {
|
||||
void Comiss(IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
// First, set flags according to Arm FCMP.
|
||||
HandleNZCVWrite();
|
||||
_FCmp(ElementSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
ComissFlags(InvalidateAF);
|
||||
ComissFlags();
|
||||
}
|
||||
|
||||
// Sets flags for a COMISS instruction
|
||||
void ComissFlags(bool InvalidateAF = false) {
|
||||
void ComissFlags() {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
@@ -2155,12 +2195,15 @@ private:
|
||||
Ref V_inv = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(V_inv);
|
||||
|
||||
if (!InvalidateAF) {
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
}
|
||||
// Intel: OF, SF, and AF set to zero
|
||||
// AMD: no mention of OF, SF and AF but actual hardware seems to always zero
|
||||
//
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
// OF and SF are zeroed:
|
||||
// _AXFLAG always produces N=0 (SF), V=0 (OF)
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
|
||||
// Convert NZCV from the Arm representation to an eXternal representation
|
||||
// that's totally not a euphemism for x86, nuh-uh. But maps to exactly we
|
||||
@@ -2372,7 +2415,7 @@ private:
|
||||
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift, bool DoubleWide = false);
|
||||
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
|
||||
@@ -603,7 +603,7 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), OpSize::i128Bit,
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, _ElementSize, Src2, Src1); });
|
||||
[this](IR::OpSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, Src2, Src1); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
@@ -865,13 +865,14 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
}
|
||||
} else if (ElementSize == OpSize::i32Bit) {
|
||||
auto GPRLow = Mask4Byte(Src.Low);
|
||||
auto GPRHigh = Mask4Byte(Src.High);
|
||||
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 4);
|
||||
Ref Fused = _VUnZip2(OpSize::i128Bit, OpSize::i16Bit, Src.Low, Src.High);
|
||||
Fused = _VUShrI(OpSize::i128Bit, OpSize::i16Bit, Fused, 15);
|
||||
auto ConstantUSHL = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_INCREMENTAL_U16_INDEX);
|
||||
Fused = _VUShl(OpSize::i128Bit, OpSize::i16Bit, Fused, ConstantUSHL, false);
|
||||
Fused = _VAddV(OpSize::i128Bit, OpSize::i16Bit, Fused);
|
||||
GPR = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Fused, 0);
|
||||
} else {
|
||||
auto GPRLow = Mask8Byte(Src.Low);
|
||||
auto GPRHigh = Mask8Byte(Src.High);
|
||||
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
|
||||
GPR = Mask4Byte(_VUnZip2(OpSize::i128Bit, OpSize::i32Bit, Src.Low, Src.High));
|
||||
}
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
|
||||
}
|
||||
@@ -885,7 +886,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
|
||||
auto Mask1Byte = [this](Ref Src, Ref VMask) {
|
||||
auto VCMP = _VCMPLTZ(OpSize::i128Bit, OpSize::i8Bit, Src);
|
||||
auto VAnd = _VAnd(OpSize::i128Bit, OpSize::i8Bit, VCMP, VMask);
|
||||
auto VAnd = _VAnd(OpSize::i128Bit, VCMP, VMask);
|
||||
|
||||
auto VAdd1 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAnd, VAnd);
|
||||
auto VAdd2 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAdd1, VAdd1);
|
||||
@@ -1469,6 +1470,35 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper) {
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
auto Acc = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = Helper(Acc.Low, Src1.Low, Src2.Low);
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
Result.High = Helper(Acc.High, Src1.High, Src2.High);
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPBUSD(OpcodeArgs, bool Saturating) {
|
||||
AVX128_VPDPImpl(Op,
|
||||
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPBUSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPWSSD(OpcodeArgs, bool Saturating) {
|
||||
AVX128_VPDPImpl(Op,
|
||||
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPWSSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is128Bit = SrcSize == OpSize::i128Bit;
|
||||
@@ -1729,8 +1759,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
|
||||
{
|
||||
// Calculate ZF first.
|
||||
auto AndLow = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
|
||||
auto AndHigh = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
|
||||
auto AndLow = _VAnd(OpSize::i128Bit, Src2.Low, Src1.Low);
|
||||
auto AndHigh = _VAnd(OpSize::i128Bit, Src2.High, Src1.High);
|
||||
|
||||
auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1);
|
||||
auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1);
|
||||
@@ -1749,8 +1779,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
|
||||
{
|
||||
// Calculate CF Second
|
||||
auto AndLow = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
|
||||
auto AndHigh = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
|
||||
auto AndLow = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low);
|
||||
auto AndHigh = _VAndn(OpSize::i128Bit, Src2.High, Src1.High);
|
||||
|
||||
auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1);
|
||||
auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1);
|
||||
@@ -1788,11 +1818,11 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
|
||||
}
|
||||
|
||||
// For 256-bit, we need to unroll. This is nontrivial.
|
||||
Ref Test1Low = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.Low, Src2.Low);
|
||||
Ref Test2Low = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
|
||||
Ref Test1Low = _VAnd(OpSize::i128Bit, Src1.Low, Src2.Low);
|
||||
Ref Test2Low = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low);
|
||||
|
||||
Ref Test1High = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.High, Src2.High);
|
||||
Ref Test2High = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
|
||||
Ref Test1High = _VAnd(OpSize::i128Bit, Src1.High, Src2.High);
|
||||
Ref Test2High = _VAndn(OpSize::i128Bit, Src2.High, Src1.High);
|
||||
|
||||
// Element size must be less than 32-bit for the sign bit tricks.
|
||||
Ref Test1Max = _VUMax(OpSize::i128Bit, OpSize::i16Bit, Test1Low, Test1High);
|
||||
@@ -2009,13 +2039,13 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
|
||||
ConstantEOR = LoadAndCacheNamedVectorConstant(
|
||||
OpSize::i128Bit, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT);
|
||||
}
|
||||
auto InvertedSourceLow = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].Low, ConstantEOR);
|
||||
auto InvertedSourceLow = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].Low, ConstantEOR);
|
||||
|
||||
Result.Low = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].Low, Sources[Src2Idx - 1].Low, InvertedSourceLow);
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
auto InvertedSourceHigh = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].High, ConstantEOR);
|
||||
auto InvertedSourceHigh = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].High, ConstantEOR);
|
||||
Result.High = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].High, Sources[Src2Idx - 1].High, InvertedSourceHigh);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
|
||||
@@ -5,21 +5,30 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
// Instructions
|
||||
{0x00, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0>},
|
||||
{0x00, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0, false>},
|
||||
{0x04, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0, true>},
|
||||
|
||||
{0x08, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0>},
|
||||
{0x08, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0, false>},
|
||||
{0x0c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0, true>},
|
||||
|
||||
{0x10, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0>},
|
||||
{0x10, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0, false>},
|
||||
{0x14, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0, true>},
|
||||
|
||||
{0x18, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0>},
|
||||
{0x18, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0, false>},
|
||||
{0x1c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0, true>},
|
||||
|
||||
{0x20, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0>},
|
||||
{0x20, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0, false>},
|
||||
{0x24, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0, true>},
|
||||
|
||||
{0x28, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0>},
|
||||
{0x28, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0, false>},
|
||||
{0x2c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0, true>},
|
||||
|
||||
{0x30, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0>},
|
||||
{0x30, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0, false>},
|
||||
{0x34, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0, true>},
|
||||
|
||||
{0x38, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0, false>},
|
||||
{0x3c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0, true>},
|
||||
|
||||
{0x38, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0>},
|
||||
{0x50, 8, &OpDispatchBuilder::PUSHREGOp},
|
||||
{0x58, 8, &OpDispatchBuilder::POPOp},
|
||||
{0x68, 1, &OpDispatchBuilder::PUSHOp},
|
||||
@@ -29,7 +38,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0x6C, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{0x70, 16, &OpDispatchBuilder::CondJUMPOp},
|
||||
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0, false>},
|
||||
{0x86, 2, &OpDispatchBuilder::XCHGOp},
|
||||
{0x88, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
|
||||
@@ -37,7 +46,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0x8D, 1, &OpDispatchBuilder::LEAOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, true>},
|
||||
{0x8F, 1, &OpDispatchBuilder::POPOp},
|
||||
{0x90, 8, &OpDispatchBuilder::XCHGOp},
|
||||
{0x90, 8, &OpDispatchBuilder::XCHGRAXOp},
|
||||
|
||||
{0x98, 1, &OpDispatchBuilder::CDQOp},
|
||||
{0x99, 1, &OpDispatchBuilder::CQOOp},
|
||||
@@ -49,7 +58,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xA4, 2, &OpDispatchBuilder::MOVSOp},
|
||||
|
||||
{0xA6, 2, &OpDispatchBuilder::CMPSOp},
|
||||
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0, true>},
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
|
||||
@@ -26,17 +26,25 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
// Move the element to zero, rotate, and then move back (Using duplicates).
|
||||
// Saves one instruction versus that path that doesn't support SHA extension.
|
||||
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Sha1HRotated = _VSha1H(Duplicated);
|
||||
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
Ref Result {};
|
||||
if (CTX->HostFeatures.SupportsSVE128) {
|
||||
auto ZeroVec = LoadZeroVector(OpSize::i128Bit);
|
||||
auto Tmp = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, ZeroVec, Dest);
|
||||
auto Xar = _VXar(OpSize::i128Bit, OpSize::i32Bit, ZeroVec, Tmp, 2);
|
||||
Result = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, Xar);
|
||||
} else {
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
// Move the element to zero, rotate, and then move back (Using duplicates).
|
||||
// Saves one instruction versus that path that doesn't support SHA extension.
|
||||
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Sha1HRotated = _VSha1H(Duplicated);
|
||||
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
}
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
@@ -50,9 +58,9 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
Ref Result = _VXor(OpSize::i128Bit, Dest, NewVec);
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -70,7 +78,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
// The result is swizzled differently than expected
|
||||
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -99,7 +107,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i128Bit);
|
||||
|
||||
Ref Src1 = SHADataShuffle(Dest);
|
||||
Ref Src2 = SHADataShuffle(Src);
|
||||
@@ -112,7 +120,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -125,7 +133,7 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VSha256U0(Dest, Src);
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
@@ -142,7 +150,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VSha256U1(Src1, Src2);
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
@@ -177,17 +185,22 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
auto Result = shuffle_abcd(A, B);
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs, bool IsAVX) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResultFPR(Op, Result);
|
||||
|
||||
if (IsAVX) {
|
||||
StoreResultFPR(Op, Result);
|
||||
} else {
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
@@ -198,19 +211,30 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
const auto Is256Bit = DstSize == OpSize::i256Bit;
|
||||
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
Ref ZeroVec = LoadZeroVector(DstSize);
|
||||
|
||||
Ref Result {};
|
||||
if (Is256Bit) {
|
||||
// TODO: Handle as one operation once vixl supports it.
|
||||
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
|
||||
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
|
||||
|
||||
auto Lower = _VAESEnc(OpSize::i128Bit, State, Key, ZeroVec);
|
||||
auto Upper = _VAESEnc(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
|
||||
|
||||
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
|
||||
} else {
|
||||
Result = _VAESEnc(DstSize, State, Key, ZeroVec);
|
||||
}
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
@@ -223,19 +247,30 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
const auto Is256Bit = DstSize == OpSize::i256Bit;
|
||||
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
Ref ZeroVec = LoadZeroVector(DstSize);
|
||||
|
||||
Ref Result {};
|
||||
if (Is256Bit) {
|
||||
// TODO: Handle as one operation once vixl supports it.
|
||||
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
|
||||
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
|
||||
|
||||
auto Lower = _VAESEncLast(OpSize::i128Bit, State, Key, ZeroVec);
|
||||
auto Upper = _VAESEncLast(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
|
||||
|
||||
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
|
||||
} else {
|
||||
Result = _VAESEncLast(DstSize, State, Key, ZeroVec);
|
||||
}
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
@@ -248,19 +283,30 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
const auto Is256Bit = DstSize == OpSize::i256Bit;
|
||||
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
Ref ZeroVec = LoadZeroVector(DstSize);
|
||||
|
||||
Ref Result {};
|
||||
if (Is256Bit) {
|
||||
// TODO: Handle as one operation once vixl supports it.
|
||||
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
|
||||
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
|
||||
|
||||
auto Lower = _VAESDec(OpSize::i128Bit, State, Key, ZeroVec);
|
||||
auto Upper = _VAESDec(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
|
||||
|
||||
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
|
||||
} else {
|
||||
Result = _VAESDec(DstSize, State, Key, ZeroVec);
|
||||
}
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
@@ -273,19 +319,30 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResultFPR(Op, Result);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
const auto Is256Bit = DstSize == OpSize::i256Bit;
|
||||
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
Ref ZeroVec = LoadZeroVector(DstSize);
|
||||
|
||||
Ref Result {};
|
||||
if (Is256Bit) {
|
||||
// TODO: Handle as one operation once vixl supports it.
|
||||
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
|
||||
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
|
||||
|
||||
auto Lower = _VAESDecLast(OpSize::i128Bit, State, Key, ZeroVec);
|
||||
auto Upper = _VAESDecLast(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
|
||||
|
||||
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
|
||||
} else {
|
||||
Result = _VAESDecLast(DstSize, State, Key, ZeroVec);
|
||||
}
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
@@ -298,14 +355,19 @@ Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(OpSize::i128Bit), RCON);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs, bool IsAVX) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResultFPR(Op, Result);
|
||||
|
||||
if (IsAVX) {
|
||||
StoreResultFPR(Op, Result);
|
||||
} else {
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -317,8 +379,8 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResultFPR(Op, Res);
|
||||
auto Result = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
|
||||
@@ -7,7 +7,7 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, false>},
|
||||
{0x1D, 1, &OpDispatchBuilder::PF2IDOp},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
|
||||
@@ -26,8 +26,8 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 2>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
|
||||
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
@@ -35,7 +35,7 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0xB0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 0>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
|
||||
{0xB7, 1, &OpDispatchBuilder::PMULHRWOp},
|
||||
|
||||
{0xBB, 1, &OpDispatchBuilder::PSWAPDOp},
|
||||
|
||||
@@ -432,7 +432,7 @@ void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res) {
|
||||
SetNZP_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift, bool DoubleWide) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -447,8 +447,12 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Re
|
||||
// Extract the last bit shifted in to CF. Shift is already masked, but for
|
||||
// 8/16-bit it might be >= SrcSizeBits, in which case CF is cleared. There's
|
||||
// nothing to do in that case since we already cleared CF above.
|
||||
//
|
||||
// - Double-wide shift has UB when shift is GREATER-THAN operand.
|
||||
// - Single-wide shift has UB when shift is GREATER-THAN-EQUAL operand.
|
||||
const auto SrcSizeBits = IR::OpSizeAsBits(SrcSize);
|
||||
if (Shift < SrcSizeBits) {
|
||||
const bool ShouldSetCF = DoubleWide ? (Shift <= SrcSizeBits) : (Shift < SrcSizeBits);
|
||||
if (ShouldSetCF) {
|
||||
SetCFDirect(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -79,7 +79,7 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESImcOp, false>},
|
||||
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
|
||||
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
|
||||
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
|
||||
|
||||
@@ -37,7 +37,7 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, false>},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESKeyGenAssist, false>},
|
||||
|
||||
};
|
||||
return std::to_array(Table);
|
||||
|
||||
@@ -9,36 +9,36 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
// GROUP 1
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>}, // CMP
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
|
||||
|
||||
// GROUP 2
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
@@ -46,8 +46,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
@@ -73,8 +73,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::RCLSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::RCRSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLSmallerOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCRSmallerOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
@@ -82,16 +82,16 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
|
||||
|
||||
// GROUP 3
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, &OpDispatchBuilder::MULOp},
|
||||
@@ -99,8 +99,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
@@ -42,6 +43,12 @@ void OpDispatchBuilder::MOVVectorUnalignedOp(OpcodeArgs) {
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVVectorUnalignedNoNopOp(OpcodeArgs) {
|
||||
// Moves to same register might have secondary-effects and can't convert to a nop.
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs, bool IsAVX) {
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
@@ -366,13 +373,13 @@ Ref OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
|
||||
auto Result = VectorScalarUnaryInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
|
||||
auto Result = VectorScalarUnaryInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
|
||||
}
|
||||
|
||||
@@ -523,14 +530,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
|
||||
case VectorCompareType::NLT_US: // NGT(Swapped operand)
|
||||
case VectorCompareType::NLT_UQ: {
|
||||
Ref Result = _VFCMPLT(ElementSize, ElementSize, Src1, Src2);
|
||||
Result = _VNot(ElementSize, ElementSize, Result);
|
||||
Result = _VNot(ElementSize, Result);
|
||||
// Insert the lower bits
|
||||
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
case VectorCompareType::NLE_US: // NGE(Swapped operand)
|
||||
case VectorCompareType::NLE_UQ: {
|
||||
Ref Result = _VFCMPLE(ElementSize, ElementSize, Src1, Src2);
|
||||
Result = _VNot(ElementSize, ElementSize, Result);
|
||||
Result = _VNot(ElementSize, Result);
|
||||
// Insert the lower bits
|
||||
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
@@ -539,14 +546,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
|
||||
case VectorCompareType::NGT_UQ:
|
||||
case VectorCompareType::NGT_US: {
|
||||
Ref Result = _VFCMPLT(ElementSize, ElementSize, Src2, Src1);
|
||||
Result = _VNot(ElementSize, ElementSize, Result);
|
||||
Result = _VNot(ElementSize, Result);
|
||||
// Insert the lower bits
|
||||
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
case VectorCompareType::NGE_UQ:
|
||||
case VectorCompareType::NGE_US: {
|
||||
Ref Result = _VFCMPLE(ElementSize, ElementSize, Src2, Src1);
|
||||
Result = _VNot(ElementSize, ElementSize, Result);
|
||||
Result = _VNot(ElementSize, Result);
|
||||
// Insert the lower bits
|
||||
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
@@ -567,10 +574,10 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
|
||||
// If either of the sources are unordered, then returns true.
|
||||
Ref Src1_U = _VFCMPEQ(Size, ElementSize, Src1, Src1);
|
||||
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
|
||||
auto Ordered = _VAnd(Size, ElementSize, Src1_U, Src2_U);
|
||||
auto Ordered = _VAnd(Size, Src1_U, Src2_U);
|
||||
|
||||
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
|
||||
Ref Result = _VOrn(Size, ElementSize, Compare_Ordered, Ordered);
|
||||
Ref Result = _VOrn(Size, Compare_Ordered, Ordered);
|
||||
|
||||
// Insert the lower bits
|
||||
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
|
||||
@@ -582,8 +589,8 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
|
||||
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
|
||||
|
||||
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
|
||||
Ref Result = _VAndn(Size, ElementSize, Src1_U, Compare_Ordered);
|
||||
Result = _VAnd(Size, ElementSize, Result, Src2_U);
|
||||
Ref Result = _VAndn(Size, Src1_U, Compare_Ordered);
|
||||
Result = _VAnd(Size, Result, Src2_U);
|
||||
|
||||
// Insert the lower bits
|
||||
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
|
||||
@@ -598,14 +605,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const uint8_t CompType = Op->Src[1].Literal();
|
||||
const uint8_t CompType = Op->Src[1].Literal() & 0b111;
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType & 0b111, false);
|
||||
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType, false);
|
||||
|
||||
// ARM doesn't have any instructions that handle the semantics of NLT and NLE directly.
|
||||
// In fact, these are the two SSE compatison types where we cannot use VFCMPScalarInsert
|
||||
@@ -623,7 +630,7 @@ void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const uint8_t CompType = Op->Src[2].Literal();
|
||||
const uint8_t CompType = Op->Src[2].Literal() & 0b11111;
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
|
||||
@@ -633,7 +640,7 @@ void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize
|
||||
Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], DstSize, Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType & 0b11111, true);
|
||||
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType, true);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
|
||||
}
|
||||
|
||||
@@ -789,7 +796,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
Ref VMask = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_MOVMASKB);
|
||||
|
||||
auto VCMP = _VCMPLTZ(SrcSize, OpSize::i8Bit, Src);
|
||||
auto VAnd = _VAnd(SrcSize, OpSize::i8Bit, VCMP, VMask);
|
||||
auto VAnd = _VAnd(SrcSize, VCMP, VMask);
|
||||
|
||||
// Since we also handle the MM MOVMSKB here too,
|
||||
// we need to clamp the lower bound.
|
||||
@@ -884,7 +891,7 @@ Ref OpDispatchBuilder::PSHUFBOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, Ref
|
||||
// the lane splitting behavior, so cap the maximum size at 16.
|
||||
const auto SanitizedSrcSize = std::min(SrcSize, OpSize::i128Bit);
|
||||
|
||||
Ref MaskedIndices = _VAnd(SrcSize, SrcSize, Src2, MaskVector);
|
||||
Ref MaskedIndices = _VAnd(SrcSize, Src2, MaskVector);
|
||||
|
||||
Ref Low = _VTBL1(SanitizedSrcSize, Src1, MaskedIndices);
|
||||
if (!Is256Bit) {
|
||||
@@ -1999,7 +2006,7 @@ void OpDispatchBuilder::VANDNOp(OpcodeArgs) {
|
||||
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Dest = _VAndn(SrcSize, SrcSize, Src2, Src1);
|
||||
Ref Dest = _VAndn(SrcSize, Src2, Src1);
|
||||
|
||||
StoreResultFPR(Op, Dest);
|
||||
}
|
||||
@@ -2581,9 +2588,10 @@ Ref OpDispatchBuilder::CVTGPR_To_FPRImpl(OpcodeArgs, IR::OpSize DstElementSize,
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
// GPR size is determined by REX.W
|
||||
// But instruction does not support 16bit register operands
|
||||
const auto GPRSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
|
||||
|
||||
if (CTX->HostFeatures.SupportsFRINTTS) {
|
||||
// When we have FRINTTS, this is a two-step process. First, we round to the
|
||||
@@ -2616,7 +2624,9 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, boo
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : SrcElementSize;
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src, SrcElementSize, HostRoundingMode);
|
||||
StoreResultGPR(Op, Result);
|
||||
|
||||
const auto DestSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, DestSize);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen) {
|
||||
@@ -2782,12 +2792,12 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSiz
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref MaskSrc = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref MaskSrc = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
// Mask only cares about the top bit of each byte
|
||||
MaskSrc = _VCMPLTZ(Size, OpSize::i8Bit, MaskSrc);
|
||||
|
||||
// Vector that will overwrite byte elements.
|
||||
Ref VectorSrc = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
Ref VectorSrc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
// RDI source (DS prefix by default)
|
||||
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
@@ -2839,11 +2849,15 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
// Loading from GPR and moving to Vector.
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
|
||||
const auto SrcSize = std::max(OpSize::i32Bit, OpSizeFromSrc(Op));
|
||||
// zext to 128bit
|
||||
Result = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
|
||||
Result = _VCastFromGPR(OpSize::i128Bit, SrcSize, Src);
|
||||
} else {
|
||||
// Loading from Memory as a scalar. Zero extend
|
||||
Result = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
const auto SrcSize = std::max(OpSize::i32Bit, OpSizeFromSrc(Op));
|
||||
Result = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
StoreResult_WithAVXInsert(VectorType, RegClass::FPR, Op, Result);
|
||||
@@ -2851,14 +2865,19 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) {
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
if (Op->Dest.IsGPR()) {
|
||||
const auto ElementSize = OpSizeFromDst(Op);
|
||||
const auto DstSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
|
||||
|
||||
// Extract element from GPR. Zero extending in the process.
|
||||
Src = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src, 0);
|
||||
Src = _VExtractToGPR(OpSizeFromSrc(Op), DstSize, Src, 0);
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize);
|
||||
} else {
|
||||
const auto DstSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
|
||||
|
||||
// Storing first element to memory.
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src, OpSize::i8Bit);
|
||||
_StoreMemFPR(DstSize, Dest, Src, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2878,24 +2897,24 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1
|
||||
case VectorCompareType::NLT_US: // NGT(Swapped operand)
|
||||
case VectorCompareType::NLT_UQ: {
|
||||
Ref Result = _VFCMPLT(Size, ElementSize, Src1, Src2);
|
||||
return _VNot(Size, ElementSize, Result);
|
||||
return _VNot(Size, Result);
|
||||
}
|
||||
case VectorCompareType::NLE_US: // NGE(Swapped operand)
|
||||
case VectorCompareType::NLE_UQ: {
|
||||
Ref Result = _VFCMPLE(Size, ElementSize, Src1, Src2);
|
||||
return _VNot(Size, ElementSize, Result);
|
||||
return _VNot(Size, Result);
|
||||
}
|
||||
case VectorCompareType::ORD_Q:
|
||||
case VectorCompareType::ORD_S: return _VFCMPORD(Size, ElementSize, Src1, Src2);
|
||||
case VectorCompareType::NGT_UQ:
|
||||
case VectorCompareType::NGT_US: {
|
||||
Ref Result = _VFCMPLT(Size, ElementSize, Src2, Src1);
|
||||
return _VNot(Size, ElementSize, Result);
|
||||
return _VNot(Size, Result);
|
||||
}
|
||||
case VectorCompareType::NGE_UQ:
|
||||
case VectorCompareType::NGE_US: {
|
||||
Ref Result = _VFCMPLE(Size, ElementSize, Src2, Src1);
|
||||
return _VNot(Size, ElementSize, Result);
|
||||
return _VNot(Size, Result);
|
||||
}
|
||||
case VectorCompareType::GT_OQ:
|
||||
case VectorCompareType::GT_OS: return _VFCMPLT(Size, ElementSize, Src2, Src1);
|
||||
@@ -2906,10 +2925,10 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1
|
||||
// If either of the sources are unordered, then returns true.
|
||||
Ref Src1_U = _VFCMPEQ(Size, ElementSize, Src1, Src1);
|
||||
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
|
||||
auto Ordered = _VAnd(Size, ElementSize, Src1_U, Src2_U);
|
||||
auto Ordered = _VAnd(Size, Src1_U, Src2_U);
|
||||
|
||||
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
|
||||
return _VOrn(Size, ElementSize, Compare_Ordered, Ordered);
|
||||
return _VOrn(Size, Compare_Ordered, Ordered);
|
||||
}
|
||||
case VectorCompareType::NEQ_OQ:
|
||||
case VectorCompareType::NEQ_OS: {
|
||||
@@ -2918,8 +2937,8 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1
|
||||
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
|
||||
|
||||
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
|
||||
Ref Result = _VAndn(Size, ElementSize, Src1_U, Compare_Ordered);
|
||||
return _VAnd(Size, ElementSize, Result, Src2_U);
|
||||
Ref Result = _VAndn(Size, Src1_U, Compare_Ordered);
|
||||
return _VAnd(Size, Result, Src2_U);
|
||||
}
|
||||
case VectorCompareType::FALSE_OQ:
|
||||
case VectorCompareType::FALSE_OS: return LoadZeroVector(Size);
|
||||
@@ -3231,9 +3250,10 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, MemBase, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref Top {};
|
||||
{
|
||||
auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -3242,10 +3262,21 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
auto MMRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 32);
|
||||
_StoreContextFPR(OpSize::i128Bit, MMRegs.Low, MMBaseOffset() + i * 16);
|
||||
_StoreContextFPR(OpSize::i128Bit, MMRegs.High, MMBaseOffset() + (i + 1) * 16);
|
||||
auto SevenConst = Constant(7);
|
||||
auto low = Constant(~0ULL);
|
||||
auto high = Constant(0xFFFF);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3289,7 +3320,7 @@ void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
|
||||
|
||||
// On top of resetting the flags to a default state, we also need to clear
|
||||
// all of the ST0-7/MM0-7 registers to zero.
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i64Bit);
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i128Bit);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
_StoreContextFPR(OpSize::i128Bit, ZeroVector, MMBaseOffset() + i * 16);
|
||||
}
|
||||
@@ -3485,7 +3516,7 @@ Ref OpDispatchBuilder::ADDSUBPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Sr
|
||||
} else {
|
||||
auto ConstantEOR =
|
||||
LoadAndCacheNamedVectorConstant(Size, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PADDSUBPS_INVERT : NAMED_VECTOR_PADDSUBPD_INVERT);
|
||||
auto InvertedSource = _VXor(Size, ElementSize, Src2, ConstantEOR);
|
||||
auto InvertedSource = _VXor(Size, Src2, ConstantEOR);
|
||||
return _VFAdd(Size, ElementSize, Src1, InvertedSource);
|
||||
}
|
||||
}
|
||||
@@ -3571,15 +3602,25 @@ void OpDispatchBuilder::PF2IWOp(OpcodeArgs) {
|
||||
// Float to int32_t
|
||||
Src = _Vector_FToZS(Size, OpSize::i32Bit, Src);
|
||||
|
||||
// We now need to transpose the lower 16-bits of each element together
|
||||
// Only needing to move the upper element down in this case
|
||||
Src = _VUnZip(Size, OpSize::i16Bit, Src, Src);
|
||||
// Truncate the 32-bit integers to 16-bit
|
||||
// Saturate values outside the 16-bit range to smallest and largest 16-bit values
|
||||
Src = _VSQXTN(Size, OpSize::i32Bit, Src);
|
||||
|
||||
// Now we need to sign extend the 16bit value to 32-bit
|
||||
Src = _VSXTL(Size, OpSize::i16Bit, Src);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src, Size);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PF2IDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
Src = _Vector_FToZS(Size, OpSize::i32Bit, Src);
|
||||
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src, Size);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PMULHRWOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
@@ -3732,6 +3773,84 @@ void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) {
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
|
||||
// Does four 8-bit unsigned * signed byte multiplies per 32-bit element, sums them and accumulates in to the destination
|
||||
|
||||
if (CTX->HostFeatures.SupportsI8MM) {
|
||||
// The I8MM extension maps onto VPDP* almost directly.
|
||||
if (!Saturating) {
|
||||
return _VUSDot(Size, Acc, Src1, Src2);
|
||||
}
|
||||
|
||||
auto DotProduct = _VUSDot(Size, LoadZeroVector(Size), Src1, Src2);
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsDotProd) {
|
||||
// VSDOT assumes signed input, so we need to convert Src1 to signed, and then
|
||||
// perform correction afterwards using 0x40 to account for the unsigned input.
|
||||
auto Src1Signed = _VXor(Size, Src1, _VectorImm(Size, OpSize::i8Bit, 0x80));
|
||||
auto SixtyFour = _VectorImm(Size, OpSize::i8Bit, 0x40);
|
||||
|
||||
auto DotProduct = _VSDot(Size, Saturating ? LoadZeroVector(Size) : Acc, Src2, SixtyFour);
|
||||
DotProduct = _VSDot(Size, DotProduct, Src2, SixtyFour);
|
||||
DotProduct = _VSDot(Size, DotProduct, Src1Signed, Src2);
|
||||
if (Saturating) {
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
return DotProduct;
|
||||
}
|
||||
|
||||
// Naive software implementation.
|
||||
auto Even = _VUnZip(Size, OpSize::i16Bit, Src1, Src2);
|
||||
auto Even1_16b = _VUXTL(Size, OpSize::i8Bit, Even);
|
||||
auto Even2_16b = _VSXTL2(Size, OpSize::i8Bit, Even);
|
||||
auto ResMul_Even = _VMul(Size, OpSize::i16Bit, Even1_16b, Even2_16b);
|
||||
|
||||
auto Odd = _VUnZip2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
auto Odd1_16b = _VUXTL(Size, OpSize::i8Bit, Odd);
|
||||
auto Odd2_16b = _VSXTL2(Size, OpSize::i8Bit, Odd);
|
||||
auto ResMul_Odd = _VMul(Size, OpSize::i16Bit, Odd1_16b, Odd2_16b);
|
||||
|
||||
auto DotProduct = _VSAdALP(Size, OpSize::i16Bit, _VSAddLP(Size, OpSize::i16Bit, ResMul_Even), ResMul_Odd);
|
||||
if (Saturating) {
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
|
||||
auto DotProduct = PMADDWDOpImpl(Size, Src1, Src2);
|
||||
if (!Saturating) {
|
||||
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
auto NegDotProduct = _VNeg(Size, OpSize::i32Bit, DotProduct);
|
||||
return _VSQSub(Size, OpSize::i32Bit, Acc, NegDotProduct);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPDPBUSDOp(OpcodeArgs, bool Saturating) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
|
||||
Ref Result = VPDPBUSDOpImpl(Size, Acc, Src1, Src2, Saturating);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPDPWSSDOp(OpcodeArgs, bool Saturating) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
|
||||
Ref Result = VPDPWSSDOpImpl(Size, Acc, Src1, Src2, Saturating);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
if (Signed) {
|
||||
@@ -3764,32 +3883,14 @@ void OpDispatchBuilder::VPMULHWOp(OpcodeArgs, bool Signed) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2) {
|
||||
Ref Res {};
|
||||
if (Size == OpSize::i64Bit) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2);
|
||||
Res = _VSShrI(Size << 1, OpSize::i32Bit, Res, 14);
|
||||
auto OneVector = _VectorImm(Size << 1, OpSize::i32Bit, 1);
|
||||
Res = _VAdd(Size << 1, OpSize::i32Bit, Res, OneVector);
|
||||
return _VUShrNI(Size << 1, OpSize::i32Bit, Res, 1);
|
||||
Ref Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2);
|
||||
return _VRSHRN(Size << 1, OpSize::i32Bit, Res, 15);
|
||||
} else {
|
||||
// 128-bit and 256-bit are less efficient
|
||||
Ref ResultLow;
|
||||
Ref ResultHigh;
|
||||
|
||||
ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2);
|
||||
ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
|
||||
ResultLow = _VSShrI(Size, OpSize::i32Bit, ResultLow, 14);
|
||||
ResultHigh = _VSShrI(Size, OpSize::i32Bit, ResultHigh, 14);
|
||||
auto OneVector = _VectorImm(Size, OpSize::i32Bit, 1);
|
||||
|
||||
ResultLow = _VAdd(Size, OpSize::i32Bit, ResultLow, OneVector);
|
||||
ResultHigh = _VAdd(Size, OpSize::i32Bit, ResultHigh, OneVector);
|
||||
|
||||
// Combine the results
|
||||
Res = _VUShrNI(Size, OpSize::i32Bit, ResultLow, 1);
|
||||
return _VUShrNI2(Size, OpSize::i32Bit, Res, ResultHigh, 1);
|
||||
Ref ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2);
|
||||
Ref ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
return _VRSHRNPair(Size, OpSize::i32Bit, ResultLow, ResultHigh, 15);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4349,8 +4450,8 @@ void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSiz
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
Ref Test1 = _VAnd(Size, OpSize::i8Bit, Dest, Src);
|
||||
Ref Test2 = _VAndn(Size, OpSize::i8Bit, Src, Dest);
|
||||
Ref Test1 = _VAnd(Size, Dest, Src);
|
||||
Ref Test2 = _VAndn(Size, Src, Dest);
|
||||
|
||||
// Element size must be less than 32-bit for the sign bit tricks.
|
||||
Test1 = _VUMaxV(Size, OpSize::i16Bit, Test1);
|
||||
@@ -4384,11 +4485,11 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref
|
||||
|
||||
Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant));
|
||||
|
||||
Ref AndTest = _VAnd(SrcSize, OpSize::i8Bit, Src2, Src1);
|
||||
Ref AndNotTest = _VAndn(SrcSize, OpSize::i8Bit, Src2, Src1);
|
||||
Ref AndTest = _VAnd(SrcSize, Src2, Src1);
|
||||
Ref AndNotTest = _VAndn(SrcSize, Src2, Src1);
|
||||
|
||||
Ref MaskedAnd = _VAnd(SrcSize, OpSize::i8Bit, AndTest, Mask);
|
||||
Ref MaskedAndNot = _VAnd(SrcSize, OpSize::i8Bit, AndNotTest, Mask);
|
||||
Ref MaskedAnd = _VAnd(SrcSize, AndTest, Mask);
|
||||
Ref MaskedAndNot = _VAnd(SrcSize, AndNotTest, Mask);
|
||||
|
||||
Ref MaxAnd = _VUMaxV(SrcSize, OpSize::i16Bit, MaskedAnd);
|
||||
Ref MaxAndNot = _VUMaxV(SrcSize, OpSize::i16Bit, MaskedAndNot);
|
||||
@@ -4491,7 +4592,7 @@ Ref OpDispatchBuilder::DPPOpImpl(IR::OpSize DstSize, Ref Src1, Ref Src2, uint8_t
|
||||
// Now mask results based on IndexMask.
|
||||
if (SrcMask != SizeMask) {
|
||||
auto InputMask = LoadAndCacheIndexedNamedVectorConstant(DstSize, NamedIndexMask, SrcMask * 16);
|
||||
Temp = _VAnd(DstSize, ElementSize, Temp, InputMask);
|
||||
Temp = _VAnd(DstSize, Temp, InputMask);
|
||||
}
|
||||
|
||||
// Now due a float reduction
|
||||
@@ -4913,7 +5014,7 @@ void OpDispatchBuilder::VPERM2Op(OpcodeArgs) {
|
||||
|
||||
Ref OpDispatchBuilder::VPERMDIndices(OpSize DstSize, Ref Indices, Ref IndexMask, Ref Repeating3210) {
|
||||
// Get rid of any junk unrelated to the relevant selector index bits (bits [2:0])
|
||||
Ref SanitizedIndices = _VAnd(DstSize, OpSize::i8Bit, Indices, IndexMask);
|
||||
Ref SanitizedIndices = _VAnd(DstSize, Indices, IndexMask);
|
||||
|
||||
// Build up the broadcasted index mask. e.g. On x86-64, the selector index
|
||||
// is always in the lower 3 bits of a 32-bit element. However, in order to
|
||||
@@ -5108,15 +5209,12 @@ void OpDispatchBuilder::VPERMQOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint64_t Selector) {
|
||||
const auto IsWordElements = ElementSize == OpSize::i16Bit;
|
||||
const auto Is256Bit = VecSize == OpSize::i256Bit;
|
||||
if (VecSize == OpSize::i256Bit) {
|
||||
return _VBlendImm(VecSize, ElementSize, Src1, Src2, Selector);
|
||||
}
|
||||
|
||||
const auto ElementsPerLane = uint32_t(IR::NumElements(OpSize::i128Bit, ElementSize));
|
||||
|
||||
// PBLENDW uses the same immediate size for 128-bit and 256-bit
|
||||
// while all the others double in size.
|
||||
const auto MaskSize = Is256Bit && !IsWordElements ? ElementsPerLane * 2 : ElementsPerLane;
|
||||
const auto Mask = (1U << MaskSize) - 1;
|
||||
const auto Mask = (1U << ElementsPerLane) - 1;
|
||||
|
||||
// Now, we determine which mask portion has the higher population count.
|
||||
// we use this to determine which source we use as the base to insert into.
|
||||
@@ -5133,7 +5231,7 @@ Ref OpDispatchBuilder::VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize,
|
||||
// In the event we tie, then we can just use Src1 and only perform incoming insertions
|
||||
// that come from Src2.
|
||||
const auto NumSrc2Bits = uint32_t(std::popcount(Selector & Mask));
|
||||
const auto NumSrc1Bits = MaskSize - NumSrc2Bits;
|
||||
const auto NumSrc1Bits = ElementsPerLane - NumSrc2Bits;
|
||||
const auto IsUsingSrc1 = NumSrc1Bits >= NumSrc2Bits;
|
||||
Ref Result = IsUsingSrc1 ? Src1 : Src2;
|
||||
Ref Source = IsUsingSrc1 ? Src2 : Src1;
|
||||
@@ -5316,7 +5414,7 @@ Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize,
|
||||
// Sanitize indices first
|
||||
const auto ShiftAmount = 0b11 >> static_cast<uint32_t>(IsPD);
|
||||
Ref IndexMask = _VectorImm(DstSize, ElementSize, ShiftAmount);
|
||||
Ref SanitizedIndices = _VAnd(DstSize, OpSize::i8Bit, Indices, IndexMask);
|
||||
Ref SanitizedIndices = _VAnd(DstSize, Indices, IndexMask);
|
||||
|
||||
Ref IndexTrn1 = _VTrn(DstSize, OpSize::i8Bit, SanitizedIndices, SanitizedIndices);
|
||||
Ref IndexTrn2 = _VTrn(DstSize, OpSize::i16Bit, IndexTrn1, IndexTrn1);
|
||||
@@ -5353,6 +5451,132 @@ void OpDispatchBuilder::VPERMILRegOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
// Computes the number of valid elements in a string for the explicit case,
|
||||
// where "valid" means that the character is not after the Nth character in the string.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXSaturateExplicitLength(OpSize ElementSize, OpSize LengthSize, Ref RawLength) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
SaveNZCV();
|
||||
_SubNZCV(LengthSize, RawLength, Constant(0));
|
||||
Ref AbsLength = _Neg(LengthSize, RawLength, CondClass::MI);
|
||||
return _Select(OpSize::i32Bit, LengthSize, CondClass::ULT, AbsLength, Constant(NumElements), AbsLength, Constant(NumElements));
|
||||
}
|
||||
|
||||
// Equal Any is a "is character in set" operation.
|
||||
// The set of characters is passed in Src1, and we then check if every
|
||||
// character in Src2 is in that set.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXEqualAny(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// First we sanitize Src1, and zero out elements after the end of the string,
|
||||
// removing them from the matching set.
|
||||
Ref Src1Element0 = _VDupElement(OpSize::i128Bit, ElementSize, Src1, 0);
|
||||
Ref SanitizedSrc1 = _VBSL(OpSize::i128Bit, Src1ValidElements, Src1, Src1Element0);
|
||||
|
||||
// Now walk through every element in Src1, broadcast it into a temporary vector,
|
||||
// and compare it against every element in Src2, then OR the results
|
||||
// together to get the final match vector.
|
||||
Ref Matches = _VCMPEQ(OpSize::i128Bit, ElementSize, Src2, Src1Element0);
|
||||
for (uint32_t i = 1; i < NumElements; i++) {
|
||||
Ref Src1Element = _VDupElement(OpSize::i128Bit, ElementSize, SanitizedSrc1, i);
|
||||
Ref ElementMatches = _VCMPEQ(OpSize::i128Bit, ElementSize, Src2, Src1Element);
|
||||
Matches = _VOr(OpSize::i128Bit, Matches, ElementMatches);
|
||||
}
|
||||
|
||||
// Handle the case where Src1 was entirely empty, and also sanitize the results
|
||||
// for any NULLs in Src2, since those don't count towards matching.
|
||||
Ref Src1NotEmpty = _VDupElement(OpSize::i128Bit, ElementSize, Src1ValidElements, 0);
|
||||
Matches = _VAnd(OpSize::i128Bit, Matches, Src1NotEmpty);
|
||||
return _VAnd(OpSize::i128Bit, Matches, Src2ValidElements);
|
||||
}
|
||||
|
||||
|
||||
// The ranges aggregation is a weird one. It checks if every character in Src2 is
|
||||
// within a character range (ie. a-z or A-Z) similar to regex.
|
||||
// Ranges are passed in Src1 as pairs of characters in adjacent lanes.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXRanges(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements, bool IsSigned) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// Signed or unsigned greater than comparison, based on the Imm8 value
|
||||
const auto GreaterThan = [&](Ref Lhs, Ref Rhs) {
|
||||
return IsSigned ? _VCMPGT(OpSize::i128Bit, ElementSize, Lhs, Rhs) : _VUCMPGT(OpSize::i128Bit, ElementSize, Lhs, Rhs);
|
||||
};
|
||||
|
||||
// First we walk through the ranges from Src1, and construct
|
||||
// temporary vectors to represent the lower and upper bounds
|
||||
// of the comparison. Then check if both comparisons are true,
|
||||
// zero out invalid lanes, and OR the result into the final result vector.
|
||||
Ref Result {};
|
||||
for (uint32_t i = 0; i < NumElements; i += 2) {
|
||||
Ref LowerBound = _VDupElement(OpSize::i128Bit, ElementSize, Src1, i);
|
||||
Ref UpperBound = _VDupElement(OpSize::i128Bit, ElementSize, Src1, i + 1);
|
||||
Ref BelowLower = GreaterThan(LowerBound, Src2);
|
||||
Ref AboveUpper = GreaterThan(Src2, UpperBound);
|
||||
|
||||
// A range is only valid if its upper bound is a valid Src1 element.
|
||||
Ref RangeValid = _VDupElement(OpSize::i128Bit, ElementSize, Src1ValidElements, i + 1);
|
||||
|
||||
// RangeValid & ~BelowLower & ~AboveUpper
|
||||
Ref InRange = _VAndn(OpSize::i128Bit, RangeValid, BelowLower);
|
||||
InRange = _VAndn(OpSize::i128Bit, InRange, AboveUpper);
|
||||
Result = Result ? _VOr(OpSize::i128Bit, Result, InRange) : InRange;
|
||||
}
|
||||
|
||||
// Invalid (ie. NULL) Src2 characters don't count towards matching.
|
||||
return _VAnd(OpSize::i128Bit, Result, Src2ValidElements);
|
||||
}
|
||||
|
||||
// This aggregation is the simplest, and is the most similar
|
||||
// to strcmp(). It Just compares lanewise which characters are
|
||||
// equal in each string.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXEqualEach(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements) {
|
||||
// Elements match when both are valid and equal, or when both are invalid.
|
||||
Ref Equal = _VCMPEQ(OpSize::i128Bit, ElementSize, Src1, Src2);
|
||||
Ref BothValid = _VAnd(OpSize::i128Bit, Src1ValidElements, Src2ValidElements);
|
||||
Ref EitherValid = _VOr(OpSize::i128Bit, Src1ValidElements, Src2ValidElements);
|
||||
Ref ValidMatches = _VAnd(OpSize::i128Bit, Equal, BothValid);
|
||||
return _VOrn(OpSize::i128Bit, ValidMatches, EitherValid);
|
||||
}
|
||||
|
||||
// EqualOrdered implements a strstr()-like semantic finding all instances
|
||||
// of the string Src1 in Src2.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXEqualOrdered(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1Length, Ref Src2Length, Ref Src1ValidElements,
|
||||
Ref Indices) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// Broadcast needle[i] into every lane of a temp vector, then compare it to
|
||||
// the haystack vector. On succesive iterations, we shift the haystack over
|
||||
// by one element, and compare it against the next needle element.
|
||||
// AND together the results of all comparisons, and what is left should
|
||||
// be a vector where the only 'true' elements are lanes where the full
|
||||
// needle was found in the haystack, or those positions after the end of the string.
|
||||
Ref Result {};
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
Ref Needle = _VDupElement(OpSize::i128Bit, ElementSize, Src1, i);
|
||||
Ref Haystack = i == 0 ? Src2 : _VExtr(OpSize::i128Bit, ElementSize, Needle, Src2, i);
|
||||
Ref ElementsEqual = _VCMPEQ(OpSize::i128Bit, ElementSize, Haystack, Needle);
|
||||
|
||||
// TODO: I think a more optimal version of this is possible, perhaps using
|
||||
// MATCH from SVE2?
|
||||
Ref ElementsValid = _VDupElement(OpSize::i128Bit, ElementSize, Src1ValidElements, i);
|
||||
Ref ElementsEqualValid = _VOrn(OpSize::i128Bit, ElementsEqual, ElementsValid);
|
||||
|
||||
// Ternary to handle the first row case
|
||||
Result = Result ? _VAnd(OpSize::i128Bit, Result, ElementsEqualValid) : ElementsEqualValid;
|
||||
}
|
||||
|
||||
// Clear positions in the result after the end of the string.
|
||||
Ref FirstClearedPosition = Add(OpSize::i32Bit, _Sub(OpSize::i32Bit, Src2Length, Src1Length), 1);
|
||||
FirstClearedPosition =
|
||||
_Select(OpSize::i32Bit, OpSize::i32Bit, CondClass::ULT, Src2Length, Constant(NumElements), FirstClearedPosition, Constant(NumElements));
|
||||
FirstClearedPosition =
|
||||
_Select(OpSize::i32Bit, OpSize::i32Bit, CondClass::EQ, Src1Length, Constant(0), Constant(NumElements), FirstClearedPosition);
|
||||
|
||||
Ref FirstClearedPositionVector = _VDupFromGPR(OpSize::i128Bit, ElementSize, FirstClearedPosition);
|
||||
Ref KeptPositions = _VCMPGT(OpSize::i128Bit, ElementSize, FirstClearedPositionVector, Indices);
|
||||
return _VAnd(OpSize::i128Bit, Result, KeptPositions);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask, bool IsAVX) {
|
||||
const uint16_t Control = Op->Src[1].Literal();
|
||||
|
||||
@@ -5365,84 +5589,166 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, OpSize::i128Bit, Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i128Bit, Op->Flags, {.Align = OpSize::i8Bit});
|
||||
|
||||
Ref IntermediateResult {};
|
||||
// Parse the immediate control bits.
|
||||
// See section 4.1 in Intel SDM for details.
|
||||
const auto Format = static_cast<CPU::SourceData>(Control & 0b11);
|
||||
const auto Aggregation = static_cast<CPU::AggregationOp>((Control >> 2) & 0b11);
|
||||
const auto Polarity = static_cast<CPU::Polarity>((Control >> 4) & 0b11);
|
||||
const bool OutputSelection = (Control & (1 << 6)) != 0;
|
||||
|
||||
// Helper constants
|
||||
const bool IsWords = Format == CPU::SourceData::U16 || Format == CPU::SourceData::S16;
|
||||
const bool IsSigned = Format == CPU::SourceData::S8 || Format == CPU::SourceData::S16;
|
||||
const auto ElementSize = IsWords ? OpSize::i16Bit : OpSize::i8Bit;
|
||||
const bool IsEqualOrdered = Aggregation == CPU::AggregationOp::EqualOrdered;
|
||||
const bool NeedsSrc1ValidElements = true;
|
||||
const bool NeedsSrc2ValidElements = !IsEqualOrdered || Polarity == CPU::Polarity::NegativeMasked;
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// Load a vector where each element contains the value of its own index
|
||||
Ref Indices = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, IsWords ? NAMED_VECTOR_INCREMENTAL_U16_INDEX : NAMED_VECTOR_INCREMENTAL_U8_INDEX);
|
||||
|
||||
// Helper lambda that computes the lowest index element set "all high" in mask.
|
||||
Ref VecNumElements {};
|
||||
const auto LowestIndex = [&](Ref Mask) {
|
||||
if (!VecNumElements) {
|
||||
VecNumElements = _VectorImm(OpSize::i128Bit, ElementSize, NumElements);
|
||||
}
|
||||
Ref Candidates = _VBSL(OpSize::i128Bit, Mask, Indices, VecNumElements);
|
||||
return _VUMinV(OpSize::i128Bit, ElementSize, Candidates);
|
||||
};
|
||||
|
||||
Ref Src1Length {};
|
||||
Ref Src2Length {};
|
||||
Ref Src2LengthVector {};
|
||||
Ref Src1ValidElementsMask {};
|
||||
|
||||
// Compute the valid element masks for the two source strings.
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
//
|
||||
// While the control bit immediate for the instruction itself is only ever 8 bits
|
||||
// in size, we use it as a 16-bit value so that we can use the 8th bit to signify
|
||||
// whether or not RAX and RDX should be interpreted as a 64-bit value.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is64Bit = SrcSize == OpSize::i64Bit;
|
||||
const auto NewControl = uint16_t(Control | (uint16_t(Is64Bit) << 8));
|
||||
const auto LengthSize = OpSizeFromSrc(Op);
|
||||
|
||||
Ref SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
Src1Length = PCMPXSTRXSaturateExplicitLength(ElementSize, LengthSize, LoadGPRRegister(X86State::REG_RAX));
|
||||
Src2Length = PCMPXSTRXSaturateExplicitLength(ElementSize, LengthSize, LoadGPRRegister(X86State::REG_RDX));
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(Src1, Src2, SrcRAX, SrcRDX, NewControl);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
|
||||
if (IsMask) {
|
||||
// For the masked variant of the instructions, if control[6] is set, then we
|
||||
// need to expand the intermediate result into a byte or word mask (depending
|
||||
// on data size specified in control[1]) along the entire length of XMM0,
|
||||
// where set bits in the intermediate result set the corresponding entry
|
||||
// in XMM0 to all 1s and unset bits set the corresponding entry to all 0s.
|
||||
//
|
||||
// If control[6] is not set, then we just store the intermediate result as-is
|
||||
// into the least significant bits of XMM0 and zero extend it.
|
||||
const auto IsExpandedMask = (Control & 0b0100'0000) != 0;
|
||||
|
||||
if (IsExpandedMask) {
|
||||
// We need to iterate over the intermediate result and
|
||||
// expand the mask into XMM0 elements.
|
||||
const auto ElementSize = 1U << (Control & 1);
|
||||
const auto NumElements = 16U >> (Control & 1);
|
||||
|
||||
Ref Result = LoadZeroVector(OpSize::i128Bit);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
Ref SignBit = _Sbfe(OpSize::i64Bit, 1, i, IntermediateResult);
|
||||
Result = _VInsGPR(OpSize::i128Bit, IR::SizeToOpSize(ElementSize), i, Result, SignBit);
|
||||
}
|
||||
|
||||
if (IsAVX) {
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
StoreXMMRegister_WithAVXInsert(VectorOpType::SSE, 0, Result);
|
||||
}
|
||||
} else {
|
||||
// We insert the intermediate result as-is.
|
||||
Ref Result = _VCastFromGPR(OpSize::i128Bit, OpSize::i16Bit, IntermediateResult);
|
||||
|
||||
if (IsAVX) {
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
StoreXMMRegister_WithAVXInsert(VectorOpType::SSE, 0, Result);
|
||||
}
|
||||
if (NeedsSrc1ValidElements) {
|
||||
Src1ValidElementsMask = _VCMPGT(OpSize::i128Bit, ElementSize, _VDupFromGPR(OpSize::i128Bit, ElementSize, Src1Length), Indices);
|
||||
}
|
||||
if (NeedsSrc2ValidElements) {
|
||||
Src2LengthVector = _VDupFromGPR(OpSize::i128Bit, ElementSize, Src2Length);
|
||||
}
|
||||
} else {
|
||||
Ref ZeroConst = Constant(0);
|
||||
// Get index of the first NUL element, to determine length of input strings.
|
||||
Ref Src1FirstNull = LowestIndex(_VCMPEQZ(OpSize::i128Bit, ElementSize, Src1));
|
||||
Ref Src2FirstNull = LowestIndex(_VCMPEQZ(OpSize::i128Bit, ElementSize, Src2));
|
||||
Src1Length = _VExtractToGPR(OpSize::i128Bit, ElementSize, Src1FirstNull, 0);
|
||||
Src2Length = _VExtractToGPR(OpSize::i128Bit, ElementSize, Src2FirstNull, 0);
|
||||
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
const auto UseMSBIndex = (Control & 0b0100'0000) != 0;
|
||||
if (NeedsSrc1ValidElements) {
|
||||
// If the index in the lane is less than the index of the first NUL, then it's a valid element.
|
||||
Src1ValidElementsMask = _VCMPGT(OpSize::i128Bit, ElementSize, _VDupElement(OpSize::i128Bit, ElementSize, Src1FirstNull, 0), Indices);
|
||||
}
|
||||
if (NeedsSrc2ValidElements) {
|
||||
Src2LengthVector = _VDupElement(OpSize::i128Bit, ElementSize, Src2FirstNull, 0);
|
||||
}
|
||||
}
|
||||
|
||||
Ref ResultNoFlags = _Bfe(OpSize::i32Bit, 16, 0, IntermediateResult);
|
||||
Ref Src1HasInvalidElements = Select01(OpSize::i32Bit, CondClass::ULT, Src1Length, Constant(NumElements));
|
||||
Ref Src2HasInvalidElements = Select01(OpSize::i32Bit, CondClass::ULT, Src2Length, Constant(NumElements));
|
||||
Ref Src2ValidElementsMask = NeedsSrc2ValidElements ? _VCMPGT(OpSize::i128Bit, ElementSize, Src2LengthVector, Indices) : nullptr;
|
||||
|
||||
Ref IfZero = Constant(16 >> (Control & 1));
|
||||
Ref IfNotZero = UseMSBIndex ? _FindMSB(IR::OpSize::i32Bit, ResultNoFlags) : _FindLSB(IR::OpSize::i32Bit, ResultNoFlags);
|
||||
Ref Result = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, ResultNoFlags, ZeroConst, IfZero, IfNotZero);
|
||||
// Perform the aggregation operations, see section 4.1.3 in the Intel SDM.
|
||||
Ref Matches {};
|
||||
switch (Aggregation) {
|
||||
case CPU::AggregationOp::EqualAny:
|
||||
Matches = PCMPXSTRXEqualAny(ElementSize, Src1, Src2, Src1ValidElementsMask, Src2ValidElementsMask);
|
||||
break;
|
||||
case CPU::AggregationOp::Ranges:
|
||||
Matches = PCMPXSTRXRanges(ElementSize, Src1, Src2, Src1ValidElementsMask, Src2ValidElementsMask, IsSigned);
|
||||
break;
|
||||
case CPU::AggregationOp::EqualEach:
|
||||
Matches = PCMPXSTRXEqualEach(ElementSize, Src1, Src2, Src1ValidElementsMask, Src2ValidElementsMask);
|
||||
break;
|
||||
case CPU::AggregationOp::EqualOrdered:
|
||||
Matches = PCMPXSTRXEqualOrdered(ElementSize, Src1, Src2, Src1Length, Src2Length, Src1ValidElementsMask, Indices);
|
||||
break;
|
||||
}
|
||||
|
||||
// Invert the matches if the polarity is negative or negative masked,
|
||||
// see section 4.1.4 in the Intel SDM.
|
||||
if (Polarity == CPU::Polarity::Negative) {
|
||||
Matches = _VNot(OpSize::i128Bit, Matches);
|
||||
} else if (Polarity == CPU::Polarity::NegativeMasked) {
|
||||
Matches = _VXor(OpSize::i128Bit, Matches, Src2ValidElementsMask);
|
||||
}
|
||||
|
||||
// Compute the flags and result
|
||||
Ref CF {};
|
||||
Ref OF {};
|
||||
|
||||
// PCMPESTRM and PCMPISTRM
|
||||
// These instructions return a mask of the matching elements.
|
||||
if (IsMask) {
|
||||
Ref Result = Matches;
|
||||
// Byte/word mask rather than a bit mask.
|
||||
if (OutputSelection) {
|
||||
// Result is already a byte/word mask, so just compute flags.
|
||||
// CF set when no elements matched
|
||||
// OF is set when the first element matched.
|
||||
Ref AnyMatch = _VExtractToGPR(OpSize::i128Bit, ElementSize, _VUMaxV(OpSize::i128Bit, ElementSize, Matches), 0);
|
||||
CF = Select01(OpSize::i32Bit, CondClass::EQ, AnyMatch, Constant(0));
|
||||
OF = _And(OpSize::i64Bit, _VExtractToGPR(OpSize::i128Bit, ElementSize, Matches, 0), Constant(1));
|
||||
|
||||
} else {
|
||||
// Convert per element vector to a bit mask with one bit per lane.
|
||||
Ref BitPositions = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_MOVMASKB);
|
||||
Ref Zero = LoadZeroVector(OpSize::i128Bit);
|
||||
|
||||
// If the elements are words, truncate to bytes first.
|
||||
Result = IsWords ? _VUnZip(OpSize::i128Bit, OpSize::i8Bit, Matches, Zero) : Matches;
|
||||
// Now repeat this pattern until we collapse each byte into a single bit.
|
||||
Result = _VAnd(OpSize::i128Bit, Result, BitPositions);
|
||||
Result = _VAddP(OpSize::i128Bit, OpSize::i8Bit, Result, Zero);
|
||||
Result = _VAddP(OpSize::i128Bit, OpSize::i8Bit, Result, Zero);
|
||||
Result = _VAddP(OpSize::i128Bit, OpSize::i8Bit, Result, Zero);
|
||||
|
||||
// Compute the flags from the bit mask in Result
|
||||
Ref BitMask = _VExtractToGPR(OpSize::i128Bit, IsWords ? OpSize::i8Bit : OpSize::i16Bit, Result, 0);
|
||||
CF = Select01(OpSize::i32Bit, CondClass::EQ, BitMask, Constant(0));
|
||||
OF = _And(OpSize::i64Bit, BitMask, Constant(1));
|
||||
}
|
||||
|
||||
StoreXMMRegister_WithAVXInsert(IsAVX ? VectorOpType::AVX : VectorOpType::SSE, 0, Result);
|
||||
|
||||
// PCMPESTRI and PCMPISTRI
|
||||
// These instructions return either the index of the first match
|
||||
// or the number of elements in the string if no match.
|
||||
} else {
|
||||
Ref LowestMatch = _VExtractToGPR(OpSize::i128Bit, ElementSize, LowestIndex(Matches), 0);
|
||||
CF = Select01(OpSize::i32Bit, CondClass::EQ, LowestMatch, Constant(NumElements));
|
||||
OF = Select01(OpSize::i32Bit, CondClass::EQ, LowestMatch, Constant(0));
|
||||
|
||||
Ref Result = LowestMatch;
|
||||
// Most significant rather than least significant index.
|
||||
if (OutputSelection) {
|
||||
// The max is also 0 when nothing matched, LowestMatch tells those apart.
|
||||
Ref Candidates = _VAnd(OpSize::i128Bit, Indices, Matches);
|
||||
Ref HighestMatchVector = _VUMaxV(OpSize::i128Bit, ElementSize, Candidates);
|
||||
Ref HighestMatch = _VExtractToGPR(OpSize::i128Bit, ElementSize, HighestMatchVector, 0);
|
||||
Result = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClass::EQ, LowestMatch, Constant(NumElements), Constant(NumElements), HighestMatch);
|
||||
}
|
||||
|
||||
// Store the result, it is already zero-extended to 64-bit implicitly.
|
||||
StoreGPRRegister(X86State::REG_RCX, Result);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags. NZCV stored in bits 28...31 like the hw op.
|
||||
SetNZCV(IntermediateResult);
|
||||
CFInverted = false;
|
||||
// SF/ZF are set when Src1/Src2 respectively have invalid elements.
|
||||
Ref SF = Src1HasInvalidElements;
|
||||
Ref ZF = Src2HasInvalidElements;
|
||||
Ref NZCV = _Orlshl(OpSize::i64Bit, _Lshl(OpSize::i64Bit, OF, Constant(28)), CF, 29);
|
||||
NZCV = _Orlshl(OpSize::i64Bit, NZCV, ZF, 30);
|
||||
NZCV = _Orlshl(OpSize::i64Bit, NZCV, SF, 31);
|
||||
SetNZCV(NZCV);
|
||||
CFInverted = true;
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -5518,7 +5824,7 @@ void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx,
|
||||
LoadAndCacheNamedVectorConstant(Size, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT);
|
||||
}
|
||||
|
||||
auto InvertedSourc = _VXor(Size, ElementSize, Sources[AddendIdx - 1], ConstantEOR);
|
||||
auto InvertedSourc = _VXor(Size, Sources[AddendIdx - 1], ConstantEOR);
|
||||
|
||||
Ref Result = _VFMLA(Size, ElementSize, Sources[Src1Idx - 1], Sources[Src2Idx - 1], InvertedSourc);
|
||||
if (!Is256Bit) {
|
||||
@@ -5665,7 +5971,7 @@ void OpDispatchBuilder::Extrq_imm(OpcodeArgs) {
|
||||
|
||||
const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth));
|
||||
const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
|
||||
Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, MaskVector);
|
||||
Result = _VAnd(OpSize::i128Bit, Result, MaskVector);
|
||||
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
@@ -5681,7 +5987,7 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) {
|
||||
Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
|
||||
|
||||
// Mask incoming source.
|
||||
Src = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, MaskVector);
|
||||
Src = _VAnd(OpSize::i64Bit, Src, MaskVector);
|
||||
|
||||
// If shifting then shift source and mask in to the correct location.
|
||||
if (Shift) {
|
||||
@@ -5689,11 +5995,8 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) {
|
||||
MaskVector = _VShlI(OpSize::i128Bit, OpSize::i64Bit, MaskVector, Shift);
|
||||
}
|
||||
|
||||
// Negate the mask.
|
||||
MaskVector = _VNot(OpSize::i64Bit, OpSize::i64Bit, MaskVector);
|
||||
|
||||
Dest = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, MaskVector);
|
||||
const Ref Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Dest, Src);
|
||||
Dest = _VAndn(OpSize::i64Bit, Dest, MaskVector);
|
||||
const Ref Result = _VOr(OpSize::i64Bit, Dest, Src);
|
||||
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
@@ -5710,15 +6013,15 @@ void OpDispatchBuilder::Extrq(OpcodeArgs) {
|
||||
};
|
||||
|
||||
// Bits[5:0] = Mask width in bits
|
||||
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, ElementMask);
|
||||
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, Src, ElementMask);
|
||||
|
||||
// Bits[13:8] = Shift right in bits
|
||||
const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask);
|
||||
const Ref ShiftBits = _VAnd(OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask);
|
||||
|
||||
// First shift in to the correct position.
|
||||
Ref Result = _VUShr(OpSize::i64Bit, OpSize::i64Bit, Dest, ShiftBits, false);
|
||||
|
||||
Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, GenerateMask(MaskWidthBits));
|
||||
Result = _VAnd(OpSize::i128Bit, Result, GenerateMask(MaskWidthBits));
|
||||
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
@@ -5737,21 +6040,19 @@ void OpDispatchBuilder::Insertq(OpcodeArgs) {
|
||||
};
|
||||
|
||||
// Bits[5:0] = Mask width in bits
|
||||
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, ElementMask);
|
||||
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, SelectorBits, ElementMask);
|
||||
|
||||
// Bits[13:8] = Shift right in bits
|
||||
const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask);
|
||||
const Ref ShiftBits = _VAnd(OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask);
|
||||
|
||||
// Extract the source data and put in to the correct location
|
||||
const Ref SrcMask = GenerateMask(MaskWidthBits);
|
||||
Ref SrcData = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Src, SrcMask);
|
||||
Ref SrcData = _VAnd(OpSize::i128Bit, Src, SrcMask);
|
||||
SrcData = _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcData, ShiftBits, false);
|
||||
|
||||
// Generate a destination mask
|
||||
const Ref DstMask = _VNot(OpSize::i64Bit, OpSize::i64Bit, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false));
|
||||
|
||||
Ref Result = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, DstMask);
|
||||
Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Result, SrcData);
|
||||
Ref Result = _VAndn(OpSize::i64Bit, Dest, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false));
|
||||
Result = _VOr(OpSize::i64Bit, Result, SrcData);
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
|
||||
}
|
||||
|
||||
|
||||
@@ -139,9 +139,7 @@ void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
|
||||
void OpDispatchBuilder::FSTToStack(OpcodeArgs) {
|
||||
const uint8_t Offset = Op->OP & 7;
|
||||
if (Offset != 0) {
|
||||
_StoreStackToStack(Offset);
|
||||
}
|
||||
_StoreStackToStack(Offset);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
@@ -172,8 +170,9 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
|
||||
|
||||
// Set Invalid Operation flag if overflow or special value
|
||||
// The x87 exception flags are sticky. Preserve earlier result
|
||||
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(InvalidFlag);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), InvalidFlag));
|
||||
}
|
||||
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
@@ -575,7 +574,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
Reg = _VAnd(OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
@@ -667,7 +666,6 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
} else {
|
||||
// OF, SF, AF, PF all undefined
|
||||
SetCFDirect(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
@@ -675,10 +673,17 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
|
||||
// Intel: OF, SF, and AF set to zero
|
||||
// AMD: no mention of OF, SF and AF but actual hardware seems to always zero
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_RAW_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
// Set Invalid Operation flag when unordered (NaN comparison)
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
|
||||
// The x87 exception flags are sticky. Preserve earlier result
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), HostFlag_Unordered));
|
||||
|
||||
if (PopTwice) {
|
||||
_PopStackDestroy();
|
||||
@@ -703,7 +708,8 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
|
||||
// Set Invalid Operation flag when unordered (NaN comparison)
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
|
||||
// The x87 exception flags are sticky. Preserve earlier result
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), HostFlag_Unordered));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2) {
|
||||
@@ -719,6 +725,9 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
|
||||
} else {
|
||||
_DecStackTop();
|
||||
}
|
||||
|
||||
// C1 set to 0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
// Operations dealing with loading and storing environment pieces
|
||||
@@ -759,10 +768,14 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
// Store Status Word
|
||||
// There's no load Status Word instruction but you can load it through frstor
|
||||
// or fldenv.
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs, bool DestRAX) {
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResultGPR(Op, StatusWord);
|
||||
if (DestRAX) {
|
||||
StoreGPRRegister(X86State::REG_RAX, StatusWord, OpSize::i16Bit);
|
||||
} else {
|
||||
StoreResultGPR(Op, StatusWord);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
@@ -847,22 +860,103 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto a = _ReadStackValue(0);
|
||||
Ref Result =
|
||||
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
Ref Value = ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
|
||||
// Extract the sign bit, which goes in C1
|
||||
Ref Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Value) : _Bfe(OpSize::i64Bit, 1, 15, Value);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
// We don't support anything else
|
||||
auto TopValid = _StackValidTag(0);
|
||||
auto NotEmpty = _StackValidTag(0);
|
||||
Ref IsEmpty = _Xor(OpSize::i64Bit, NotEmpty, Constant(1));
|
||||
Ref IsNaN {};
|
||||
Ref IsDenormal {};
|
||||
Ref IsInf {};
|
||||
Ref IsZero {};
|
||||
Ref IsUnsupported {};
|
||||
Ref NoSignBit {};
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
|
||||
// TODO: The codegen for this is not optimal, and can probably be improved
|
||||
// if FXAM ends up on the hot path for some workload.
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
constexpr uint64_t ExponentMask = 0x7FF0'0000'0000'0000ULL;
|
||||
NoSignBit = _Bfe(OpSize::i64Bit, 63, 0, Value);
|
||||
IsInf = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(ExponentMask));
|
||||
IsNaN = Select01(OpSize::i64Bit, CondClass::UGT, NoSignBit, Constant(ExponentMask));
|
||||
|
||||
IsZero = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(0));
|
||||
// 64 bit floats can't represent an x87 denormal, nor any of the
|
||||
// unsupported encodings.
|
||||
IsDenormal = Constant(0);
|
||||
IsUnsupported = Constant(0);
|
||||
} else {
|
||||
Ref Mantissa = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 0);
|
||||
|
||||
// "J" is the name given to the msb of the mantissa in the SDM.
|
||||
Ref JBit = _Bfe(OpSize::i64Bit, 1, 63, Mantissa);
|
||||
Ref Exponent = _Bfe(OpSize::i64Bit, 15, 0, Value);
|
||||
Ref IsExponentZero = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0));
|
||||
Ref IsExponentMax = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0x7FFF));
|
||||
|
||||
// Inf is when mantissa only has the J bit set, exponent is all 1's.
|
||||
Ref IsOnlyJBit = Select01(OpSize::i64Bit, CondClass::EQ, Mantissa, Constant(1ULL << 63));
|
||||
IsInf = _And(OpSize::i64Bit, IsExponentMax, IsOnlyJBit);
|
||||
|
||||
// NaN is when the low 63 bits of the mantissa are non-zero
|
||||
// and exponent is max, and the J bit is set.
|
||||
Ref Fraction = _Bfe(OpSize::i64Bit, 63, 0, Mantissa);
|
||||
Ref FractionNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Fraction, Constant(0));
|
||||
Ref IsExponentMaxWithJBit = _And(OpSize::i64Bit, IsExponentMax, JBit);
|
||||
IsNaN = _And(OpSize::i64Bit, IsExponentMaxWithJBit, FractionNonZero);
|
||||
|
||||
// Zero and Denormal are basically the same as the 64-bit case.
|
||||
Ref MantissaNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Mantissa, Constant(0));
|
||||
Ref MantissaZero = _Xor(OpSize::i64Bit, MantissaNonZero, Constant(1));
|
||||
IsZero = _And(OpSize::i64Bit, IsExponentZero, MantissaZero);
|
||||
IsDenormal = _And(OpSize::i64Bit, IsExponentZero, MantissaNonZero);
|
||||
|
||||
// This is where things are weird. If the J bit is not set
|
||||
// and the exponent is non-zero, then this is an "unsupported"
|
||||
// encoding, which I believe is left in for legacy reasons.
|
||||
Ref IsSupported = _Or(OpSize::i64Bit, IsExponentZero, JBit);
|
||||
IsUnsupported = _Xor(OpSize::i64Bit, IsSupported, Constant(1));
|
||||
}
|
||||
|
||||
// NormalFiniteNumber = !Zero && !Denormal && !Inf && !NaN && !Empty && !Unsupported
|
||||
Ref temp1 = _Or(OpSize::i64Bit, IsZero, IsDenormal);
|
||||
Ref temp2 = _Or(OpSize::i64Bit, IsInf, IsNaN);
|
||||
Ref temp3 = _Or(OpSize::i64Bit, IsUnsupported, IsEmpty);
|
||||
temp1 = _Or(OpSize::i64Bit, temp1, temp2);
|
||||
temp2 = _Or(OpSize::i64Bit, temp1, temp3);
|
||||
Ref NormalFiniteNumber = _Xor(OpSize::i64Bit, temp2, Constant(1));
|
||||
|
||||
// Set C3, C2, C0 based on the class of the FP value
|
||||
// Table is from "FXAM" page in the SDM.
|
||||
// +----------------------+----+----+----+
|
||||
// | Class | C3 | C2 | C0 |
|
||||
// +----------------------+----+----+----+
|
||||
// | Unsupported | 0 | 0 | 0 |
|
||||
// | NaN | 0 | 0 | 1 |
|
||||
// | Normal finite number | 0 | 1 | 0 |
|
||||
// | Infinity | 0 | 1 | 1 |
|
||||
// | Zero | 1 | 0 | 0 |
|
||||
// | Empty | 1 | 0 | 1 |
|
||||
// | Denormal number | 1 | 1 | 0 |
|
||||
// +----------------------+----+----+----+
|
||||
|
||||
// C0 = IsNaN || IsInf || IsEmpty
|
||||
Ref C0 = _Or(OpSize::i64Bit, IsNaN, IsInf);
|
||||
C0 = _Or(OpSize::i64Bit, C0, IsEmpty);
|
||||
|
||||
// C2 = (IsInf || Denormal || NormalFiniteNumber) && !IsEmpty
|
||||
Ref C2 = _Or(OpSize::i64Bit, IsInf, IsDenormal);
|
||||
C2 = _Or(OpSize::i64Bit, C2, NormalFiniteNumber);
|
||||
C2 = _And(OpSize::i64Bit, C2, NotEmpty);
|
||||
|
||||
// C3 = Zero || IsEmpty || Denormal
|
||||
Ref C3 = _Or(OpSize::i64Bit, IsZero, IsEmpty);
|
||||
C3 = _Or(OpSize::i64Bit, C3, IsDenormal);
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C0);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(C2);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
|
||||
|
||||
@@ -367,7 +367,7 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
} else {
|
||||
HandleNZCVWrite();
|
||||
_F80CmpValue(b);
|
||||
ComissFlags(true /* InvalidateAF */);
|
||||
ComissFlags();
|
||||
}
|
||||
|
||||
if (PopTwice) {
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/SharedCodeBufferManager.h"
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size, bool ShouldBeNamed)
|
||||
: AllocatedSize(Size) {
|
||||
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
if (ShouldBeNamed) {
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", Ptr, Size);
|
||||
}
|
||||
|
||||
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
|
||||
FEXCore::Allocator::VirtualTHPControl(Ptr, Size, FEXCore::Allocator::THPControl::Enable);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
|
||||
CodeBufferEnd = Ptr + UsableSize();
|
||||
CodeBufferOffset = Ptr;
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
|
||||
}
|
||||
|
||||
SharedCodeBufferManager::SharedCodeBufferManager() {
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
|
||||
// Only name the JIT buffers if perf JIT naming is disabled.
|
||||
// `perf top` prefers VMA names over the JIT symbols file for some reason.
|
||||
// Breaks memory tracking when naming is enabled, but it's a debug feature so it isn't expected to be enabled by default.
|
||||
NameJITBuffers = !(GlobalJITNaming || LibraryJITNaming || BlockJITNaming);
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::AllocateNew(size_t Size) {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
//
|
||||
// MDWE prevents applications from creating RWX memory mappings.
|
||||
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
|
||||
//
|
||||
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
|
||||
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
|
||||
//
|
||||
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
|
||||
//
|
||||
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
|
||||
// -1: The kernel doesn't support MDWE
|
||||
// 0: MDWE is supported but disabled
|
||||
// >0: MDWE is enabled, hence prohibiting RWX mappings
|
||||
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
|
||||
if (MDWE != -1 && MDWE != 0) {
|
||||
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
|
||||
}
|
||||
#endif
|
||||
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size, NameJITBuffers);
|
||||
|
||||
Latest = Buffer;
|
||||
|
||||
OnCodeBufferAllocated(Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::GetLatest() {
|
||||
if (!Latest) {
|
||||
AllocateNew(INITIAL_CODE_SIZE);
|
||||
}
|
||||
return Latest;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartLargerCodeBuffer() {
|
||||
if (!Latest) {
|
||||
// Allocate initial CodeBuffer and return it
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->TotalAllocationSize();
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartMaximalCodeBuffer() {
|
||||
return AllocateNew(MAX_CODE_SIZE);
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -0,0 +1,138 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
category: code buffer ~ Thread shared code buffer management
|
||||
tags: backend|shared
|
||||
$end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestToHostMap;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
struct CodeBuffer {
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size, bool ShouldBeNamed);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
CodeBuffer(CodeBuffer&& oth) = delete;
|
||||
CodeBuffer& operator=(CodeBuffer&&) = delete;
|
||||
|
||||
~CodeBuffer();
|
||||
|
||||
// Atomically allocate a fixed size buffer out of the current allocated codebuffer.
|
||||
// Lockless because it's just a linear allocator.
|
||||
struct CodeBufferAllocation {
|
||||
const uint8_t* BufferBase;
|
||||
uint8_t* BufferAllocationOffset;
|
||||
};
|
||||
|
||||
CodeBufferAllocation AtomicAllocateBuffer(size_t Size) {
|
||||
Size = FEXCore::AlignUp(Size, 16);
|
||||
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBufferOffset.load()) % 16 == 0, "Buffer needs to always be 16B aligned!");
|
||||
|
||||
auto ExpectedOffset = CodeBufferOffset.load(std::memory_order_relaxed);
|
||||
auto DesiredOffset = ExpectedOffset + Size;
|
||||
|
||||
if (DesiredOffset > CodeBufferEnd) {
|
||||
// Couldn't fit.
|
||||
return {};
|
||||
}
|
||||
|
||||
while (!CodeBufferOffset.compare_exchange_strong(ExpectedOffset, DesiredOffset)) {
|
||||
DesiredOffset = ExpectedOffset + Size;
|
||||
|
||||
if (DesiredOffset > CodeBufferEnd) {
|
||||
// Couldn't fit.
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
// Managed to fit.
|
||||
return {
|
||||
.BufferBase = Ptr,
|
||||
.BufferAllocationOffset = ExpectedOffset,
|
||||
};
|
||||
}
|
||||
|
||||
// Returns the total number of bytes available for storing code
|
||||
size_t UsableSize() const {
|
||||
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
// Returns the full size of the buffer, including the guard page.
|
||||
size_t TotalAllocationSize() const {
|
||||
return AllocatedSize;
|
||||
}
|
||||
|
||||
// Returns the num of bytes currently allocated from the allocator.
|
||||
size_t AllocatedSpaceUsed() const {
|
||||
return CodeBufferOffset - Ptr;
|
||||
}
|
||||
|
||||
// Trivially reset the allocator.
|
||||
void Reset() {
|
||||
CodeBufferOffset = Ptr;
|
||||
}
|
||||
|
||||
// Returns the base of the buffer.
|
||||
uint8_t* GetBufferBase() const {
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
private:
|
||||
uint8_t* Ptr;
|
||||
uint8_t* CodeBufferEnd;
|
||||
size_t AllocatedSize; // including guard page; see UsableSize()
|
||||
|
||||
// Code buffer allocation information.
|
||||
std::atomic<uint8_t*> CodeBufferOffset {};
|
||||
};
|
||||
|
||||
/**
|
||||
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
|
||||
*
|
||||
* The CodeBuffer is managed as a partially persistent data structure:
|
||||
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
|
||||
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
|
||||
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
|
||||
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
|
||||
*/
|
||||
class SharedCodeBufferManager {
|
||||
public:
|
||||
SharedCodeBufferManager();
|
||||
virtual ~SharedCodeBufferManager() = default;
|
||||
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
|
||||
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
|
||||
|
||||
// Allocate a new CodeBuffer with maximum internal size.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartMaximalCodeBuffer();
|
||||
|
||||
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
|
||||
bool NameJITBuffers {true};
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -83,22 +83,22 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
},
|
||||
// ENTRY_27
|
||||
{
|
||||
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
|
||||
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_2F
|
||||
{
|
||||
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
|
||||
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_37
|
||||
{
|
||||
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
|
||||
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_3F
|
||||
{
|
||||
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
|
||||
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_40
|
||||
@@ -134,23 +134,23 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
},
|
||||
// ENTRY_A0
|
||||
{
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A1
|
||||
{
|
||||
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A2
|
||||
{
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A3
|
||||
{
|
||||
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_CE
|
||||
{
|
||||
@@ -159,17 +159,17 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
},
|
||||
// ENTRY_D4
|
||||
{
|
||||
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
|
||||
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D5
|
||||
{
|
||||
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
|
||||
{"REX2", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D6
|
||||
{
|
||||
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
|
||||
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_EA
|
||||
@@ -204,8 +204,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x02, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x03, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0x06, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_06] }}},
|
||||
{0x07, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_07] }}},
|
||||
@@ -214,16 +214,16 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x0A, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x0B, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x0E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_0E] }}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x12, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x13, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x16, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_16] }}},
|
||||
{0x17, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_17] }}},
|
||||
|
||||
@@ -231,8 +231,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x1A, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x1B, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x1E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1E] }}},
|
||||
{0x1F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1F] }}},
|
||||
|
||||
@@ -240,32 +240,32 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x22, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x23, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0x27, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_27] }}},
|
||||
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x2A, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x2B, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x2F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_2F] }}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x32, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x33, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0x37, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_37] }}},
|
||||
{0x38, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x39, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x3A, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x3B, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x3F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_3F] }}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_40] }}},
|
||||
@@ -318,12 +318,12 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x8A, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8B, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x8C, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM, 0}},
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{0x8E, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8F, 1, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_ZERO_REG | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_SF_DST_RDX | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0}},
|
||||
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_NONE, 0}},
|
||||
|
||||
// These three are all X87 instructions
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0}},
|
||||
@@ -343,17 +343,17 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0xC2, 1, X86InstInfo{"RET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xC3, 1, X86InstInfo{"RET", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0}},
|
||||
@@ -362,7 +362,7 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xCA, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xCB, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 1}},
|
||||
{0xCE, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_CE] }}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
|
||||
@@ -371,9 +371,9 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xD6, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_D6] }}},
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
|
||||
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
|
||||
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
|
||||
{0xE3, 1, X86InstInfo{"JrCXZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
|
||||
// Should just throw GP
|
||||
|
||||
@@ -36,11 +36,11 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> PrimaryGroup_ArchSelect_LUT = {{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1> }},
|
||||
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1, false> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1> }},
|
||||
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1, false> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
@@ -56,7 +56,7 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> PrimaryGroup_ArchSelect_LUT = {{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::CMPOp, 1> }},
|
||||
{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::CMPOp, 1, false> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
}};
|
||||
@@ -75,14 +75,14 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 6), 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 7), 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
// Duplicates the 0x80 opcode group
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 0), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_0] }}},
|
||||
@@ -140,23 +140,23 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD1), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, X86InstInfo{"ROR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, X86InstInfo{"RCL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, X86InstInfo{"RCR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, X86InstInfo{"SHR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, X86InstInfo{"SAR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, X86InstInfo{"ROR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, X86InstInfo{"RCL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, X86InstInfo{"RCR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, X86InstInfo{"SHR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, X86InstInfo{"SAR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, X86InstInfo{"ROL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, X86InstInfo{"ROR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, X86InstInfo{"RCL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, X86InstInfo{"RCR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, X86InstInfo{"SHR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, X86InstInfo{"ROL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, X86InstInfo{"ROR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, X86InstInfo{"RCL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, X86InstInfo{"RCR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, X86InstInfo{"SHR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
// GROUP 3
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
@@ -168,7 +168,7 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, X86InstInfo{"DIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, X86InstInfo{"IDIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, X86InstInfo{"NOT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, X86InstInfo{"NEG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
@@ -196,7 +196,7 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT, 1}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 7), 1, X86InstInfo{"XABORT", TYPE_INST, FLAGS_MODRM, 1}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 7), 1, X86InstInfo{"XBEGIN", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_SETS_RIP | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
};
|
||||
|
||||
@@ -50,7 +50,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableO
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_SRC_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
|
||||
@@ -26,8 +26,8 @@ enum Secondary_LUT {
|
||||
|
||||
constexpr std::array<X86InstInfo[2], ENTRY_MAX> Secondary_ArchSelect_LUT = {{
|
||||
{
|
||||
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
|
||||
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
|
||||
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
|
||||
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
|
||||
},
|
||||
{
|
||||
{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX> } },
|
||||
@@ -204,17 +204,17 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
|
||||
{0xA0, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A0] }}},
|
||||
{0xA1, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A1] }}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA6, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA8, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A8] }}},
|
||||
{0xA9, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A9] }}},
|
||||
{0xAA, 1, X86InstInfo{"RSM", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xAC, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
|
||||
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAE, 1, X86InstInfo{"", TYPE_GROUP_15, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAF, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
|
||||
@@ -303,7 +303,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0x3E, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, sizeof(IR::SHA256Sum)}},
|
||||
#endif
|
||||
};
|
||||
|
||||
|
||||
@@ -312,6 +312,11 @@ namespace AVX128 {
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, false>},
|
||||
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, true>},
|
||||
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, false>},
|
||||
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i128Bit>},
|
||||
@@ -757,6 +762,11 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, false>},
|
||||
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, true>},
|
||||
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, false>},
|
||||
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i128Bit>},
|
||||
@@ -808,7 +818,7 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0xB6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, true, 2, 3, 1>}, // VFMADDSUB
|
||||
{OPD(2, 0b01, 0xB7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, false, 2, 3, 1>}, // VFMSUBADD
|
||||
|
||||
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESImcOp, true>},
|
||||
{OPD(2, 0b01, 0xDC), 1, &OpDispatchBuilder::VAESEncOp},
|
||||
{OPD(2, 0b01, 0xDD), 1, &OpDispatchBuilder::VAESEncLastOp},
|
||||
{OPD(2, 0b01, 0xDE), 1, &OpDispatchBuilder::VAESDecOp},
|
||||
@@ -860,7 +870,7 @@ namespace AVX256 {
|
||||
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRMOp, true>},
|
||||
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, true>},
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESKeyGenAssist, true>},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
@@ -1207,6 +1217,11 @@ auto BaseTableLambda = [](const auto RuntimeTable) consteval {
|
||||
{OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, X86InstInfo{"VPDPBUSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x51), 1, X86InstInfo{"VPDPBUSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x52), 1, X86InstInfo{"VPDPWSSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x53), 1, X86InstInfo{"VPDPWSSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_L_1 | FLAGS_SF_MOD_MEM_ONLY | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
@@ -128,6 +128,7 @@ struct DecodedOperand {
|
||||
RIPRelativeRelocation,
|
||||
Literal,
|
||||
LiteralRelocation,
|
||||
LiteralPatchable,
|
||||
SIB,
|
||||
SIBRelocation
|
||||
};
|
||||
@@ -159,6 +160,9 @@ struct DecodedOperand {
|
||||
bool IsLiteralRelocation() const {
|
||||
return Type == OpType::LiteralRelocation;
|
||||
}
|
||||
bool IsLiteralPatchable() const {
|
||||
return Type == OpType::LiteralPatchable;
|
||||
}
|
||||
bool IsSIB() const {
|
||||
return Type == OpType::SIB;
|
||||
}
|
||||
@@ -167,7 +171,7 @@ struct DecodedOperand {
|
||||
}
|
||||
|
||||
uint64_t Literal() const {
|
||||
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
|
||||
LOGMAN_THROW_A_FMT(IsLiteral() || IsLiteralPatchable(), "Precondition: must be a literal");
|
||||
return Data.Literal.Value;
|
||||
}
|
||||
|
||||
@@ -181,10 +185,14 @@ struct DecodedOperand {
|
||||
struct {
|
||||
int64_t Displacement;
|
||||
uint8_t GPR;
|
||||
bool PatchableDisp;
|
||||
uint8_t DispOffset;
|
||||
} GPRIndirect; // Shared with GPRIndirectRelocation
|
||||
|
||||
struct {
|
||||
int64_t Value;
|
||||
bool PatchableDisp;
|
||||
uint8_t DispOffset;
|
||||
} RIPLiteral; // Shared with RIPLiteralRelocation
|
||||
|
||||
struct LiteralType {
|
||||
@@ -196,12 +204,20 @@ struct DecodedOperand {
|
||||
int64_t EntrypointOffset;
|
||||
} LiteralRelocation;
|
||||
|
||||
struct {
|
||||
uint64_t Value;
|
||||
uint8_t Size;
|
||||
uint8_t FieldOffset;
|
||||
uint8_t Width;
|
||||
} LiteralPatchable;
|
||||
struct {
|
||||
int64_t Offset;
|
||||
uint8_t Scale;
|
||||
uint8_t Index; // ~0 invalid
|
||||
uint8_t Base; // ~0 invalid
|
||||
} SIB; // Shared with SIBRelocation
|
||||
bool PatchableDisp;
|
||||
uint8_t DispOffset;
|
||||
} SIB; // Shared with SIBRelocation
|
||||
};
|
||||
|
||||
TypeUnion Data;
|
||||
@@ -339,11 +355,8 @@ namespace InstFlags {
|
||||
constexpr InstFlagType FLAGS_X87_FLAGS = (1ULL << 10);
|
||||
|
||||
// Non-XMM subflags
|
||||
constexpr InstFlagType FLAGS_SF_DST_RAX = (1ULL << 11);
|
||||
constexpr InstFlagType FLAGS_SF_DST_RDX = (1ULL << 12);
|
||||
constexpr InstFlagType FLAGS_SF_SRC_RAX = (1ULL << 13);
|
||||
constexpr InstFlagType FLAGS_SF_SRC_RCX = (1ULL << 14);
|
||||
constexpr InstFlagType FLAGS_SF_REX_IN_BYTE = (1ULL << 15);
|
||||
constexpr InstFlagType FLAGS_SF_REX_IN_BYTE = (1ULL << 11);
|
||||
// subflag [15:12] unused
|
||||
|
||||
// XMM subflags
|
||||
constexpr InstFlagType FLAGS_SF_UNUSED = (1ULL << 11); // No assigned behavior yet
|
||||
@@ -393,7 +406,8 @@ namespace InstFlags {
|
||||
|
||||
constexpr InstFlagType FLAGS_CALL = (1ULL << 30);
|
||||
constexpr InstFlagType FLAGS_SUPPORTS_LOCK = (1ULL << 31);
|
||||
// Flags [57..32]: Undefined
|
||||
constexpr InstFlagType FLAGS_LITERAL_PATCHABLE = (1ULL << 32);
|
||||
// Flags [57..33]: Undefined
|
||||
// Flags [60..58]: Dst size
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
// Flags [63..61]: Src size
|
||||
@@ -409,13 +423,6 @@ namespace InstFlags {
|
||||
constexpr InstFlagType SIZE_256BIT = 0b110;
|
||||
constexpr InstFlagType SIZE_64BITDEF = 0b111; // Default mode is 64bit instead of typical 32bit
|
||||
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY;
|
||||
#else
|
||||
// Syscall ends a block on WIN32 because the instruction can update the CPU's RIP.
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY | FLAGS_BLOCK_END;
|
||||
#endif
|
||||
|
||||
constexpr InstFlagType GetSizeDstFlags(InstFlagType Flags) {
|
||||
return (Flags >> FLAGS_SIZE_DST_OFF) & SIZE_MASK;
|
||||
}
|
||||
|
||||
@@ -216,7 +216,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F64OpTable = {{
|
||||
// 5 = Invalid
|
||||
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVE},
|
||||
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, false>},
|
||||
|
||||
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
|
||||
{OPD(0xDD, 0xC8), 8, &OpDispatchBuilder::FXCH},
|
||||
@@ -284,7 +284,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F64OpTable = {{
|
||||
{OPD(0xDF, 0xD0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
{OPD(0xDF, 0xD8), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, true>},
|
||||
{OPD(0xDF, 0xE8), 8,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::FCOMIF64, OpSize::f80Bit, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDF, 0xF0), 8,
|
||||
@@ -483,7 +483,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F80OpTable = {{
|
||||
// 5 = Invalid
|
||||
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVE},
|
||||
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, false>},
|
||||
|
||||
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
|
||||
{OPD(0xDD, 0xC8), 8, &OpDispatchBuilder::FXCH},
|
||||
@@ -545,7 +545,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F80OpTable = {{
|
||||
{OPD(0xDF, 0xD0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
{OPD(0xDF, 0xD8), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, true>},
|
||||
{OPD(0xDF, 0xE8), 8,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::FCOMI, OpSize::f80Bit, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDF, 0xF0), 8,
|
||||
@@ -794,7 +794,7 @@ auto GenerateX87TableLambda = [](const auto DispatchTable) consteval {
|
||||
// / 3
|
||||
{OPD(0xDF, 0xD8), 8, X86InstInfo{"FSTP", TYPE_X87, FLAGS_SF_MOD_DST | FLAGS_POP, 0}},
|
||||
// / 4
|
||||
{OPD(0xDF, 0xE0), 1, X86InstInfo{"FNSTSW", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0}},
|
||||
{OPD(0xDF, 0xE0), 1, X86InstInfo{"FNSTSW", TYPE_INST, GenFlagsSameSize(SIZE_16BIT), 0}},
|
||||
{OPD(0xDF, 0xE1), 7, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
// / 5
|
||||
{OPD(0xDF, 0xE8), 8, X86InstInfo{"FUCOMIP", TYPE_INST, FLAGS_POP, 0}},
|
||||
|
||||
@@ -643,7 +643,7 @@ public:
|
||||
auto IROp = Node.GetNode(BaseList)->Op(IRList);
|
||||
|
||||
if (IROp->Op == OP_BEGINBLOCK) {
|
||||
auto BeginBlock = IROp->C<IROp_EndBlock>();
|
||||
auto BeginBlock = IROp->C<IROp_BeginBlock>();
|
||||
|
||||
Node = BeginBlock->BlockHeader;
|
||||
} else if (IROp->Op == OP_CODEBLOCK) {
|
||||
@@ -675,7 +675,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
[[nodiscard]]
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR);
|
||||
void Dump(fextl::ostringstream* out, const IRListView* IR);
|
||||
|
||||
constexpr auto format_as(FEXCore::IR::NodeID ID) {
|
||||
return ID.Value;
|
||||
|
||||
@@ -195,13 +195,13 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"GPR = ValidateCode Array16:$CodeOriginal, GPR:$Address, u8:$CodeLength": {
|
||||
"GPR = ValidateCode GPR:$crc, GPR:$Address, u8:$CodeLength": {
|
||||
"HasSideEffects": true,
|
||||
"HasDest": true,
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"ThreadRemoveCodeEntry": {
|
||||
"ThreadRemoveCodeEntry GPR:$EntryToInvalidate, GPR:$NewRIP": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
@@ -275,7 +275,8 @@
|
||||
"The boolean argument asks if we should be reading the reseeded number or not",
|
||||
"Reseeded RNG calculation is more expensive and will be heavier to use",
|
||||
"Returns the 64-bit number",
|
||||
"Sets the Z flag if the number is valid.",
|
||||
"Falls back to a host RNG call when the hardware doesn't support it",
|
||||
"Sets the Z flag if the number is invalid.",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
@@ -312,8 +313,8 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock, i64:$PatchSiteAddress{0}, i64:$PatchSiteSize{0}": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP - optionally patchable from guest bytes for caching"
|
||||
],
|
||||
"Inline": ["Any"],
|
||||
"HasSideEffects": true,
|
||||
@@ -326,11 +327,10 @@
|
||||
"CallbackReturn": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5": {
|
||||
"Syscall": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Dispatches a guest syscall through to the SyscallHandler class"
|
||||
],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"Thunk GPR:$ArgPtr, SHA256Sum:$ThunkNameHash": {
|
||||
@@ -954,6 +954,27 @@
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = PatchableGuestData OpSize:#Size, i64:$Value, i64:$SiteAddress, i64:$SiteSize": {
|
||||
"Desc": ["Loads Value in a patchable way",
|
||||
"On disk cache load the value is patched from live guest bytes at SiteAddress"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = PatchableGuestRIP OpSize:#Size, i64:$Value, i64:$SiteAddress, i64:$SiteSize": {
|
||||
"Desc": ["Loads GuestRIP-relative Value in a patchable way",
|
||||
"On disk cache load the value is patched from live guest PC and live displacement at SiteAddress"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = PatchableGuestCRC OpSize:#Size, i64:$Value, i64:$GuestAddress, i64:$GuestSize": {
|
||||
"Desc": ["Loads Guest CRC in a patchable way",
|
||||
"On disk cache load the value is recomputed from live guest bytes and patched"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = Constant i64:$Constant, ConstPad:$Pad{IR::ConstPad::NoPad}, i32:$MaxBytes{0}": {
|
||||
"Desc": ["Generates a 64bit constant inside of a GPR",
|
||||
"Unsupported to create a constant in FPR"
|
||||
@@ -1864,9 +1885,9 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VNot OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"FPR = VNot OpSize:#RegisterSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "OpSize::i8Bit"
|
||||
},
|
||||
|
||||
"FPR = VAbs OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
@@ -2004,15 +2025,6 @@
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
"FPR = VUShraI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -2025,7 +2037,7 @@
|
||||
|
||||
"FPR = VUShrNI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
|
||||
"Desc": ["Unsigned shifts right each element and then narrows to the next lower element size"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
@@ -2046,8 +2058,32 @@
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VRSHRN OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": ["Rounding shift right each element and then narrows to the next lower element size",
|
||||
"Writes result to the bottom half of the destination register, upper half is zeroed"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VRSHRNPair OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
|
||||
"Desc": ["Rounding shift right and narrow a pair of vectors into one result"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Sign extends elements from the source element size to the next size up",
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
@@ -2059,7 +2095,7 @@
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSSHLL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift{0}": {
|
||||
"Desc": "Sign extends elements from the source element size to the next size up",
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
@@ -2071,7 +2107,7 @@
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Zero extends elements from the source element size to the next size up",
|
||||
"Desc": ["Zero extends elements from the source element size to the next size up"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
@@ -2153,43 +2189,55 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VAnd OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"ElementSize": "OpSize::i8Bit",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VAndn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"ElementSize": "OpSize::i8Bit",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VOrn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VOrn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"ElementSize": "OpSize::i8Bit",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VOr OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"ElementSize": "OpSize::i8Bit",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VXor OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i8Bit",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VXar OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$LHS, FPR:$RHS, u8:$Rotate": {
|
||||
"Desc": [
|
||||
"Performs an XOR of corresponding elements and then rotates them right by",
|
||||
"an amount between [1, ElementSize]"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
"RegisterSize == IR::OpSize::i256Bit || RegisterSize == IR::OpSize::i128Bit"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -2214,7 +2262,7 @@
|
||||
},
|
||||
|
||||
"FPR = VAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"Desc": "Does a horizontal pairwise add of elements across the two source vectors",
|
||||
"Desc": ["Does a horizontal pairwise add of elements across the two source vectors"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2271,7 +2319,7 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"Desc": "Does a horizontal pairwise add of elements across the two source vectors with float element types",
|
||||
"Desc": ["Does a horizontal pairwise add of elements across the two source vectors with float element types"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2321,29 +2369,64 @@
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Multiplies the high elements with size extension",
|
||||
"Desc": ["Multiplies the high elements with size extension"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Multiplies the high elements with size extension",
|
||||
"Desc": ["Multiplies the high elements with size extension"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned by signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
|
||||
"Requires FEAT_I8MM."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i32Bit",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
|
||||
"Requires FEAT_DotProd."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i32Bit",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VSAddLP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Signed add long pairwise. Adds adjacent pairs of elements in to elements of twice the size.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSAdALP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Acc, FPR:$Vector": {
|
||||
"Desc": ["Signed add and accumulate long pairwise. Adds adjacent pairs of elements in to the elements of Acc, which are twice the size.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Wide unsigned multiply returning the high results",
|
||||
"Desc": ["Wide unsigned multiply returning the high results"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Wide signed multiply returning the high results",
|
||||
|
||||
"Desc": ["Wide signed multiply returning the high results"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUABDL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned Absolute Difference Long"
|
||||
],
|
||||
"Desc": ["Unsigned Absolute Difference Long"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
@@ -2435,6 +2518,14 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUCMPGT OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Vector compare unsigned greater than",
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
|
||||
],
|
||||
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPEQ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
@@ -2502,7 +2593,8 @@
|
||||
},
|
||||
|
||||
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"Desc": ["NOTE: Currently unused. The OpcodeDispatcher implements the SSE4.2 string instructions inline.",
|
||||
"Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
|
||||
@@ -2514,7 +2606,8 @@
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
|
||||
"Desc": ["NOTE: Currently unused. The OpcodeDispatcher implements the SSE4.2 string instructions inline.",
|
||||
"Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
|
||||
"This will return the intermediate result of a PCMPISTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
|
||||
@@ -2564,6 +2657,24 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VBlendImm OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$LHS, FPR:$RHS, u16:$Selector": {
|
||||
"Desc": [
|
||||
"Functions the same way an immediate blend operation on x86 would.",
|
||||
"That is: (e.g. using 16-bit elements)",
|
||||
" if (Selector[0] == 1)",
|
||||
" Dst[15:0] = RHS[15:0]",
|
||||
" else",
|
||||
" Dst[15:0] = LHS[15:0]",
|
||||
" <etc for the rest of the elements along the vector>",
|
||||
"",
|
||||
"Note that like x86, due to the selector size, the operation of this IR op",
|
||||
"uses a 128-bit lane granularity, so each blending selector independently operates",
|
||||
"on each 128-bit element that composes the vector."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
}
|
||||
},
|
||||
"Conv": {
|
||||
@@ -2601,7 +2712,7 @@
|
||||
},
|
||||
|
||||
"FPR = Vector_SToF OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Vector op: Converts signed integer to same size float",
|
||||
"Desc": ["Vector op: Converts signed integer to same size float"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2613,12 +2724,12 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToZS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Vector op: Converts float to signed integer, rounding towards zero",
|
||||
"Desc": ["Vector op: Converts float to signed integer, rounding towards zero"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToF OpSize:#RegisterSize, OpSize:#DestElementSize, FPR:$Vector, OpSize:$SrcElementSize": {
|
||||
"Desc": "Vector op: Converts float from source element size to destination size (fp32<->fp64)",
|
||||
"Desc": ["Vector op: Converts float from source element size to destination size (fp32<->fp64)"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "DestElementSize"
|
||||
},
|
||||
@@ -2673,75 +2784,74 @@
|
||||
},
|
||||
"Crypto": {
|
||||
"FPR = VAESImc FPR:$Vector": {
|
||||
"Desc": "Does a stage of the inverse mix column transformation",
|
||||
"Desc": ["Does a stage of the inverse mix column transformation"],
|
||||
"DestSize": "OpSize::i128Bit"
|
||||
},
|
||||
"FPR = VAESEnc OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
|
||||
"Desc": "Does a step of AES encryption",
|
||||
"Desc": ["Does a step of AES encryption"],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VAESEncLast OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
|
||||
"Desc": "Does the last step of AES encryption",
|
||||
"Desc": ["Does the last step of AES encryption"],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VAESDec OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
|
||||
"Desc": "Does a step of AES decryption",
|
||||
"Desc": ["Does a step of AES decryption"],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VAESDecLast OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
|
||||
"Desc": "Does the last step of AES decryption",
|
||||
"Desc": ["Does the last step of AES decryption"],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VAESKeyGenAssist FPR:$Src, FPR:$KeyGenTBLSwizzle, FPR:$ZeroReg, u8:$RCON": {
|
||||
"Desc": "Assists in key generation",
|
||||
"Desc": ["Assists in key generation"],
|
||||
"DestSize": "OpSize::i128Bit"
|
||||
},
|
||||
"FPR = VSha1H FPR:$Src": {
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"Desc": ["Does vector scalar SHA1H instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i32Bit"
|
||||
},
|
||||
"FPR = VSha1C FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1C instruction",
|
||||
"Desc": ["Does vector SHA1C instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1M FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1M instruction",
|
||||
"Desc": ["Does vector SHA1M instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1P FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1P instruction",
|
||||
"Desc": ["Does vector SHA1P instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1SU1 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"Desc": ["Does vector scalar SHA1H instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U0 instruction",
|
||||
"Desc": ["Does vector scalar VSha256U0 instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256U1 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U1 instruction",
|
||||
"Desc": ["Does vector scalar VSha256U1 instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit"
|
||||
},
|
||||
"FPR = VSha256H FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector scalar VSha256H instruction",
|
||||
"Desc": ["Does vector scalar VSha256H instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256H2 FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector scalar VSha256H2 instruction",
|
||||
"Desc": ["Does vector scalar VSha256H2 instruction"],
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, OpSize:$SrcSize": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"],
|
||||
"DestSize": "OpSize::i32Bit"
|
||||
},
|
||||
"FPR = PCLMUL OpSize:#RegisterSize, FPR:$Src1, FPR:$Src2, u8:$Selector": {
|
||||
|
||||
@@ -30,19 +30,19 @@ namespace FEXCore::IR {
|
||||
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, const SHA256Sum& Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, const SHA256Sum& Arg) {
|
||||
*out << fextl::fmt::format("sha256:{:02x}", fmt::join(Arg.data, ""));
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, uint64_t Arg) {
|
||||
*out << fextl::fmt::format("#{:#x}", Arg);
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, const char* const Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, const char* const Arg) {
|
||||
*out << fextl::fmt::format("'{}'", Arg);
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, CondClass Arg) {
|
||||
if (Arg == CondClass::AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
@@ -55,7 +55,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg)
|
||||
*out << CondNames[FEXCore::ToUnderlying(Arg)];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, MemOffsetType Arg) {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
@@ -65,7 +65,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType
|
||||
*out << Names[FEXCore::ToUnderlying(Arg)];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, RegClass Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case RegClass::Invalid: return "Invalid";
|
||||
@@ -79,7 +79,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg)
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsImmediate()) {
|
||||
auto PhyReg = PhysicalRegister(Arg);
|
||||
|
||||
@@ -128,7 +128,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, FenceType Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case FenceType::Load: return "Loads";
|
||||
@@ -140,7 +140,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg)
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, RoundMode Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case RoundMode::Nearest: return "Nearest";
|
||||
@@ -153,7 +153,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg)
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, ConstPad Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, ConstPad Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case ConstPad::NoPad: return "NoPad";
|
||||
@@ -164,7 +164,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, ConstPad Arg)
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorConstant Arg) {
|
||||
*out << [Arg] {
|
||||
// clang-format off
|
||||
switch (Arg) {
|
||||
@@ -172,6 +172,8 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorCon
|
||||
return "u16_incremental_index";
|
||||
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER:
|
||||
return "u16_incremental_index_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U8_INDEX:
|
||||
return "u8_incremental_index";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT:
|
||||
return "addsubps_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT_UPPER:
|
||||
@@ -260,7 +262,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorCon
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVectorConstant Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, IndexNamedVectorConstant Arg) {
|
||||
*out << [Arg] {
|
||||
// clang-format off
|
||||
switch (Arg) {
|
||||
@@ -286,7 +288,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVect
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, OpSize Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case OpSize::iUnsized: return "Unsized";
|
||||
@@ -303,7 +305,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, FloatCompareOp Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: return "FEQ";
|
||||
@@ -317,14 +319,14 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
*out << static_cast<uint32_t>(Arg.TrapNumber) << ".";
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, ShiftType Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: return "LSL";
|
||||
@@ -336,7 +338,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg)
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg) {
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, BranchHint Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case BranchHint::None: return "None";
|
||||
@@ -348,11 +350,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
|
||||
*out << fextl::fmt::format("{:02x}", fmt::join(Arg, ""));
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
void Dump(fextl::ostringstream* out, const IRListView* IR) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
|
||||
@@ -19,6 +19,8 @@ namespace FEXCore::IR {
|
||||
|
||||
static bool IsFragmentExit(FEXCore::IR::IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_THREADREMOVECODEENTRY:
|
||||
case OP_SYSCALL:
|
||||
case OP_EXITFUNCTION:
|
||||
case OP_BREAK: return true;
|
||||
default: return false;
|
||||
@@ -33,7 +35,7 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
|
||||
}
|
||||
}
|
||||
|
||||
RegClass IREmitter::WalkFindRegClass(Ref Node) {
|
||||
RegClass IREmitter::WalkFindRegClass(Ref Node) const {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case RegClass::GPR:
|
||||
@@ -45,9 +47,8 @@ RegClass IREmitter::WalkFindRegClass(Ref Node) {
|
||||
}
|
||||
|
||||
// Complex case, needs to be handled on an op by op basis
|
||||
uintptr_t DataBegin = DualListData.DataBegin();
|
||||
|
||||
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
|
||||
const uintptr_t DataBegin = DualListData.DataBegin();
|
||||
const auto* IROp = Node->Op(DataBegin);
|
||||
|
||||
switch (IROp->Op) {
|
||||
case IROps::OP_LOADREGISTER: {
|
||||
|
||||
@@ -36,6 +36,10 @@ public:
|
||||
DualListData.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
DualListData.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
IRListView ViewIR() {
|
||||
return IRListView(&DualListData);
|
||||
}
|
||||
@@ -45,7 +49,7 @@ public:
|
||||
*
|
||||
* @{ */
|
||||
|
||||
RegClass WalkFindRegClass(Ref Node);
|
||||
RegClass WalkFindRegClass(Ref Node) const;
|
||||
|
||||
// These inlining helpers are used by IRDefines.inc so define first.
|
||||
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
|
||||
@@ -310,14 +314,14 @@ public:
|
||||
}
|
||||
|
||||
/** @} */
|
||||
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) const {
|
||||
auto RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) {
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) const {
|
||||
auto RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
const auto* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
if (Constant) {
|
||||
@@ -328,9 +332,9 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
bool IsValueInlineConstant(OrderedNodeWrapper ssa) {
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
bool IsValueInlineConstant(OrderedNodeWrapper ssa) const {
|
||||
auto RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
const auto* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_INLINECONSTANT) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -129,6 +129,10 @@ public:
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
PoolObject.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
private:
|
||||
Utils::PoolBufferWithTimedRetirement<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
@@ -66,25 +67,42 @@ void PassManager::Finalize() {
|
||||
}
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
void PassManager::AddDefaultPasses(Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
// We only specifically disable optimization passes if desired, as IR output should
|
||||
// still be well-formed regardless of the modifications made to it.
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultValidationPasses() {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
InsertValidationPass(Validation::CreateIRValidation(), "IRValidation");
|
||||
#endif
|
||||
}
|
||||
|
||||
void PassManager::InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
|
||||
Pass* PassManager::InsertPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name) {
|
||||
auto* PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
|
||||
AttemptNameMapping(Name, PassPtr);
|
||||
return PassPtr;
|
||||
}
|
||||
|
||||
PassManager::PassArrayType::iterator PassManager::InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass) {
|
||||
Pass->RegisterPassManager(this);
|
||||
return Passes.insert(pos, std::move(Pass));
|
||||
}
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
void PassManager::InsertValidationPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name) {
|
||||
Pass->RegisterPassManager(this);
|
||||
auto* PassPtr = ValidationPasses.emplace_back(std::move(Pass)).get();
|
||||
AttemptNameMapping(Name, PassPtr);
|
||||
}
|
||||
#endif
|
||||
|
||||
void PassManager::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
@@ -98,4 +116,15 @@ void PassManager::Run(IREmitter* IREmit) {
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void PassManager::AttemptNameMapping(const fextl::string& Name, Pass* NewPass) {
|
||||
if (Name.empty()) {
|
||||
// Empty name is a 'don't care' case. e.g. Passes that just need to run,
|
||||
// but don't need to be actively looked up.
|
||||
return;
|
||||
}
|
||||
|
||||
const auto Result = NameToPassMaping.emplace(Name, NewPass);
|
||||
LOGMAN_THROW_A_FMT(Result.second, "Tried to insert pass with name '{}'. But name is already used", Name);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -8,23 +8,18 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <functional>
|
||||
#include <concepts>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
class SyscallHandler;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class PassManager;
|
||||
class IREmitter;
|
||||
@@ -44,64 +39,57 @@ protected:
|
||||
|
||||
class PassManager final {
|
||||
public:
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl* ctx);
|
||||
void AddDefaultValidationPasses();
|
||||
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
|
||||
auto PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
|
||||
|
||||
if (!Name.empty()) {
|
||||
NameToPassMaping[Name] = PassPtr;
|
||||
}
|
||||
return PassPtr;
|
||||
explicit PassManager(Context::ContextImpl* CTX) {
|
||||
AddDefaultPasses(CTX);
|
||||
}
|
||||
|
||||
void InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
// Executes all of the passes added to the manager.
|
||||
// If assertions are enabled, this will also run all validation passes.
|
||||
void Run(IREmitter* IREmit);
|
||||
|
||||
bool HasPass(fextl::string Name) const {
|
||||
// Inserts a new pass into the manager, optionally also assigning a name to it
|
||||
// for use in the lookup functions,
|
||||
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name = "");
|
||||
|
||||
// Whether or not a pass with the given name is within the manager.
|
||||
bool HasPass(const fextl::string& Name) const {
|
||||
return NameToPassMaping.contains(Name);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T* GetPass(fextl::string Name) {
|
||||
return dynamic_cast<T*>(NameToPassMaping[Name]);
|
||||
// Retrieves a pass from the manager that has the given name assigned to it.
|
||||
// Will return nullptr if the pass doesn't exist.
|
||||
template<std::derived_from<Pass> T>
|
||||
T* GetPass(const fextl::string& Name) {
|
||||
return dynamic_cast<T*>(GetPass(Name));
|
||||
}
|
||||
|
||||
Pass* GetPass(fextl::string Name) {
|
||||
return NameToPassMaping[Name];
|
||||
}
|
||||
|
||||
void RegisterSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) {
|
||||
SyscallHandler = Handler;
|
||||
Pass* GetPass(const fextl::string& Name) {
|
||||
const auto Iter = NameToPassMaping.find(Name);
|
||||
if (Iter == NameToPassMaping.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
return Iter->second;
|
||||
}
|
||||
|
||||
// Finalizes the pass manager state and assumes no other passes will be added after called.
|
||||
// This will reorganize the execution order of the passes if necessary.
|
||||
void Finalize();
|
||||
|
||||
protected:
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
|
||||
private:
|
||||
void AddDefaultPasses(Context::ContextImpl* ctx);
|
||||
|
||||
using PassArrayType = fextl::vector<fextl::unique_ptr<Pass>>;
|
||||
PassArrayType::iterator InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass) {
|
||||
Pass->RegisterPassManager(this);
|
||||
return Passes.insert(pos, std::move(Pass));
|
||||
}
|
||||
PassArrayType::iterator InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass);
|
||||
|
||||
PassArrayType Passes;
|
||||
fextl::unordered_map<fextl::string, Pass*> NameToPassMaping;
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
fextl::vector<fextl::unique_ptr<Pass>> ValidationPasses;
|
||||
void InsertValidationPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
|
||||
Pass->RegisterPassManager(this);
|
||||
auto PassPtr = ValidationPasses.emplace_back(std::move(Pass)).get();
|
||||
|
||||
if (!Name.empty()) {
|
||||
NameToPassMaping[Name] = PassPtr;
|
||||
}
|
||||
}
|
||||
void InsertValidationPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name = "");
|
||||
#endif
|
||||
|
||||
void AttemptNameMapping(const fextl::string& Name, Pass* NewPass);
|
||||
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(PassManagerDumpIR, PASSMANAGERDUMPIR);
|
||||
};
|
||||
|
||||
@@ -57,7 +57,7 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
if (FD.IsValid() || DumpToLog) {
|
||||
fextl::stringstream out;
|
||||
fextl::ostringstream out;
|
||||
FEXCore::IR::Dump(&out, &IR);
|
||||
if (FD.IsValid()) {
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", +HeaderOp->OriginalRIP, out.str());
|
||||
|
||||
@@ -10,6 +10,7 @@ $end_info$
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/Passes/IRValidation.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
@@ -46,7 +47,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
OffsetToBlockMap.clear();
|
||||
EntryBlock = nullptr;
|
||||
|
||||
uint32_t Count = CurrentIR.GetSSACount();
|
||||
const auto Count = CurrentIR.GetSSACount();
|
||||
if (Count > MaxNodes) {
|
||||
NodeIsLive.Realloc(Count);
|
||||
}
|
||||
@@ -59,7 +60,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
#endif
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
if (!EntryBlock) {
|
||||
@@ -77,15 +78,15 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
HadError |= OpSize == IR::OpSize::iInvalid;
|
||||
// Does the op have a destination of size 0?
|
||||
// Does the op have an unsized destination?
|
||||
if (OpSize == IR::OpSize::iInvalid) {
|
||||
HadError = true;
|
||||
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
|
||||
}
|
||||
|
||||
// Does the node have zero uses? Should have been DCE'd
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
HadWarning |= true;
|
||||
HadWarning = true;
|
||||
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
@@ -98,27 +99,26 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::RegClass::Invalid) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
}
|
||||
|
||||
// If no physical register was assigned
|
||||
if (PhyReg.IsInvalid()) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
}
|
||||
|
||||
// Assigned class wasn't the expected class and it is a non-complex op
|
||||
if (AssignedClass != ExpectedClass && ExpectedClass != IR::RegClass::Complex) {
|
||||
HadWarning |= true;
|
||||
HadWarning = true;
|
||||
Warnings << "%" << ID << ": Destination had register class " << uint32_t(AssignedClass) << " When register class "
|
||||
<< uint32_t(ExpectedClass) << " Was expected" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
|
||||
|
||||
const uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
OrderedNodeWrapper Arg = IROp->Args[i];
|
||||
const auto ArgID = Arg.ID();
|
||||
@@ -126,8 +126,6 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
continue;
|
||||
}
|
||||
|
||||
IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
Uses[ArgID.Value]++;
|
||||
}
|
||||
@@ -135,10 +133,11 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
// We do not validate the location of inline constants because it's
|
||||
// irrelevant, they're ignored by RA and always inlined to where they
|
||||
// need to be. This lets us pool inline constants globally.
|
||||
bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
|
||||
const IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
|
||||
const bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
|
||||
|
||||
if (!Ignore && ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] references invalid %" << ArgID << std::endl;
|
||||
}
|
||||
}
|
||||
@@ -147,7 +146,6 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
switch (IROp->Op) {
|
||||
case IR::OP_EXITFUNCTION: {
|
||||
CurrentBlock->HasExit = true;
|
||||
break;
|
||||
}
|
||||
case IR::OP_CONDJUMP: {
|
||||
@@ -163,7 +161,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
const FEXCore::IR::IROp_Header* FalseTargetOp = CurrentIR.GetOp<IROp_Header>(FalseTargetNode);
|
||||
|
||||
if (TrueTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "CondJump %" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
} else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first;
|
||||
@@ -171,7 +169,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
if (FalseTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "CondJump %" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
} else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first;
|
||||
@@ -187,7 +185,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
const FEXCore::IR::IROp_Header* TargetOp = CurrentIR.GetOp<IROp_Header>(TargetNode);
|
||||
if (TargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "Jump %" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
} else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
|
||||
@@ -204,7 +202,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
// Blocks can only have zero (Exit), 1 (Unconditional branch) or 2 (Conditional) successors
|
||||
size_t NumSuccessors = CurrentBlock->Successors.size();
|
||||
if (NumSuccessors > 2) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
}
|
||||
|
||||
@@ -220,7 +218,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
{
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (Op != IR::OP_ENDBLOCK) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
}
|
||||
}
|
||||
@@ -231,7 +229,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
{
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (!IsBlockExit(Op)) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
}
|
||||
}
|
||||
@@ -243,7 +241,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
HadError = true;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
}
|
||||
}
|
||||
@@ -251,7 +249,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
HadWarning = false;
|
||||
if (HadError || HadWarning) {
|
||||
fextl::stringstream Out;
|
||||
fextl::ostringstream Out;
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR);
|
||||
|
||||
if (HadError) {
|
||||
|
||||
@@ -8,20 +8,16 @@
|
||||
|
||||
namespace FEXCore::IR::Validation {
|
||||
|
||||
struct BlockInfo {
|
||||
bool HasExit;
|
||||
const OrderedNode* BlockNode;
|
||||
|
||||
fextl::vector<OrderedNode*> Predecessors;
|
||||
fextl::vector<OrderedNode*> Successors;
|
||||
};
|
||||
|
||||
class IRValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~IRValidation();
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
struct BlockInfo {
|
||||
fextl::vector<OrderedNode*> Predecessors;
|
||||
fextl::vector<OrderedNode*> Successors;
|
||||
};
|
||||
|
||||
BitSet<uint64_t> NodeIsLive {};
|
||||
OrderedNode* EntryBlock {};
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -196,7 +197,7 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClass Cond)
|
||||
}
|
||||
}
|
||||
|
||||
constexpr FlagInfo ClassifyConst(IROps Op) {
|
||||
static constexpr FlagInfo ClassifyConst(IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_ANDWITHFLAGS:
|
||||
return FlagInfo::Pack({
|
||||
@@ -332,15 +333,15 @@ constexpr FlagInfo ClassifyConst(IROps Op) {
|
||||
}
|
||||
}
|
||||
|
||||
constexpr auto FlagInfos = std::invoke([] {
|
||||
constexpr auto FlagInfos = [] {
|
||||
std::array<FlagInfo, OP_LAST> ret = {};
|
||||
|
||||
for (unsigned i = 0; i < OP_LAST; ++i) {
|
||||
ret[i] = ClassifyConst((IROps)i);
|
||||
ret[i] = ClassifyConst(IROps(i));
|
||||
}
|
||||
|
||||
return ret;
|
||||
});
|
||||
}();
|
||||
|
||||
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
FlagInfo Info = FlagInfos[IROp->Op];
|
||||
@@ -351,22 +352,22 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTINCREMENT: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NZCVSELECTV: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelectV>();
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectV>();
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NEG: {
|
||||
auto Op = IROp->CW<IR::IROp_Neg>();
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
if (!Op->FromNZCV) {
|
||||
return FlagInfo::Pack({});
|
||||
}
|
||||
@@ -376,7 +377,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
|
||||
case OP_CONDSUBNZCV:
|
||||
case OP_CONDADDNZCV: {
|
||||
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
return FlagInfo::Pack({
|
||||
.Read = FlagsForCondClassType(Op->Cond),
|
||||
.Write = FLAG_NZCV,
|
||||
@@ -385,7 +386,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
}
|
||||
|
||||
case OP_RMIFNZCV: {
|
||||
auto Op = IROp->CW<IR::IROp_RmifNZCV>();
|
||||
auto Op = IROp->C<IR::IROp_RmifNZCV>();
|
||||
|
||||
static_assert(FLAG_N == (1 << 3), "rmif mask lines up with our bits");
|
||||
static_assert(FLAG_Z == (1 << 2), "rmif mask lines up with our bits");
|
||||
@@ -399,7 +400,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
}
|
||||
|
||||
case OP_INVALIDATEFLAGS: {
|
||||
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
|
||||
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
|
||||
unsigned Flags = 0;
|
||||
|
||||
// TODO: Make this translation less silly
|
||||
@@ -536,7 +537,7 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie
|
||||
// Initialize the FlagsRead mask according to the exit instruction.
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
auto Op = ExitOp->C<IR::IROp_CondJump>();
|
||||
FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
FlagsRead = CFG.Get(ExitOp->Args[0])->Flags;
|
||||
@@ -643,7 +644,7 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) {
|
||||
if (IROp->Op == OP_STOREPF) {
|
||||
auto Op = IROp->CW<IR::IROp_StorePF>();
|
||||
auto Op = IROp->C<IR::IROp_StorePF>();
|
||||
auto Generator = CurrentIR.GetOp<IR::IROp_Header>(Op->Value);
|
||||
|
||||
// Determine if we only write 0/1 to the parity flag.
|
||||
@@ -696,7 +697,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
--CodeLast;
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
auto Op = ExitOp->C<IR::IROp_CondJump>();
|
||||
|
||||
CFG.RecordEdge(Block->ID, Op->TrueBlock);
|
||||
CFG.RecordEdge(Block->ID, Op->FalseBlock);
|
||||
|
||||
@@ -33,7 +33,7 @@ namespace {
|
||||
Ref RegToSSA[32];
|
||||
};
|
||||
|
||||
IR::RegClass GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
IR::RegClass GetRegClassFromNode(const IR::IROp_Header* IROp) {
|
||||
const auto Class = IR::GetRegClass(IROp->Op);
|
||||
if (Class != IR::RegClass::Complex) {
|
||||
return Class;
|
||||
@@ -49,9 +49,14 @@ namespace {
|
||||
case IR::OP_FILLREGISTER: return IROp->C<IR::IROp_FillRegister>()->Class;
|
||||
default: return IR::RegClass::Invalid;
|
||||
}
|
||||
};
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
void RegisterAllocationPass::SetNumPairRegs(uint32_t NumRegs) {
|
||||
LOGMAN_THROW_A_FMT((NumRegs % 2) == 0, "Number of pair regs must be even. (Given: {})", NumRegs);
|
||||
PairRegs = NumRegs;
|
||||
}
|
||||
|
||||
class ConstrainedRAPass final : public RegisterAllocationPass {
|
||||
public:
|
||||
explicit ConstrainedRAPass(const FEXCore::CPUIDEmu* CPUID)
|
||||
@@ -85,27 +90,27 @@ private:
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
int64_t SourceIndex {};
|
||||
|
||||
bool Rematerializable(IROp_Header* IROp) {
|
||||
static bool Rematerializable(const IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT;
|
||||
}
|
||||
|
||||
Ref InsertFill(Ref Node) {
|
||||
IROp_Header* IROp = IR->GetOp<IROp_Header>(Node);
|
||||
const auto* IROp = IR->GetOp<IROp_Header>(Node);
|
||||
|
||||
// Remat if we can
|
||||
if (Rematerializable(IROp)) {
|
||||
const auto Op = IROp->C<IR::IROp_Constant>();
|
||||
uint64_t Const = Op->Constant;
|
||||
const auto* Op = IROp->C<IR::IROp_Constant>();
|
||||
const uint64_t Const = Op->Constant;
|
||||
return IREmit->_Constant(Const, Op->Pad, Op->MaxBytes);
|
||||
}
|
||||
|
||||
// Otherwise fill from stack
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
|
||||
const uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
|
||||
|
||||
const auto RegClass = GetRegClassFromNode(IR, IROp);
|
||||
const auto RegClass = GetRegClassFromNode(IROp);
|
||||
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
|
||||
};
|
||||
}
|
||||
|
||||
// IP of next-use of each source. IPs are measured from the end of the
|
||||
// block, so we don't need to size the block up-front.
|
||||
@@ -113,32 +118,35 @@ private:
|
||||
|
||||
bool AnySpilled {};
|
||||
|
||||
bool IsValidArg(OrderedNodeWrapper Arg) {
|
||||
bool IsValidArg(OrderedNodeWrapper Arg) const {
|
||||
if (Arg.IsInvalid()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Op = IR->GetOp<IROp_Header>(Arg)->Op;
|
||||
return Op != OP_INLINECONSTANT && Op != OP_INLINEENTRYPOINTOFFSET;
|
||||
};
|
||||
}
|
||||
|
||||
RegisterClassData* GetClass(PhysicalRegister Reg) {
|
||||
return &Classes[Reg.Class];
|
||||
};
|
||||
}
|
||||
const RegisterClassData* GetClass(PhysicalRegister Reg) const {
|
||||
return &Classes[Reg.Class];
|
||||
}
|
||||
|
||||
uint32_t GetRegBits(PhysicalRegister Reg) {
|
||||
return 1 << Reg.Reg;
|
||||
};
|
||||
static uint32_t GetRegBits(PhysicalRegister Reg) {
|
||||
return 1U << Reg.Reg;
|
||||
}
|
||||
|
||||
bool IsInRegisterFile(Ref Node) {
|
||||
bool IsInRegisterFile(Ref Node) const {
|
||||
auto ID = IR->GetID(Node).Value;
|
||||
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[ID];
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
const PhysicalRegister Reg = SSAToReg[ID];
|
||||
const RegisterClassData* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
|
||||
};
|
||||
}
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
@@ -147,7 +155,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
|
||||
Class->Available |= RegBits;
|
||||
};
|
||||
}
|
||||
|
||||
bool HasSource(IROp_Header* I, PhysicalRegister Reg) {
|
||||
int NumArgs = IR::GetRAArgs(I->Op);
|
||||
@@ -170,13 +178,13 @@ private:
|
||||
}
|
||||
|
||||
return false;
|
||||
};
|
||||
}
|
||||
|
||||
Ref DecodeSRANode(const IROp_Header* IROp, Ref Node) {
|
||||
if (IROp->Op == OP_LOADREGISTER || IROp->Op == OP_LOADPF || IROp->Op == OP_LOADAF) {
|
||||
return Node;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto V = IROp->C<IR::IROp_StorePF>()->Value;
|
||||
auto V = IROp->C<IR::IROp_StoreRegister>()->Value;
|
||||
V.ClearKill();
|
||||
return IR->GetNode(V);
|
||||
} else if (IROp->Op == OP_STOREPF || IROp->Op == OP_STOREAF) {
|
||||
@@ -186,9 +194,9 @@ private:
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
};
|
||||
}
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) const {
|
||||
uint8_t FlagOffset = Classes[FEXCore::ToUnderlying(RegClass::GPRFixed)].Count - 2;
|
||||
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
@@ -207,9 +215,9 @@ private:
|
||||
return PhysicalRegister {RegClass::GPRFixed, uint8_t(Op->Reg)};
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
bool IsTrivial(Ref Node, const IROp_Header* Header) {
|
||||
bool IsTrivial(Ref Node, const IROp_Header* Header) const {
|
||||
switch (Header->Op) {
|
||||
case OP_ALLOCATEGPR: return true;
|
||||
case OP_ALLOCATEGPRAFTER: return true;
|
||||
@@ -320,7 +328,7 @@ private:
|
||||
// If we already spilled the Candidate, we don't need to spill again.
|
||||
// Similarly, if we can rematerialize the instruction, we don't spill it.
|
||||
if (!Spilled && Header->Op != OP_CONSTANT) {
|
||||
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(IR, Header), "Consistent");
|
||||
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(Header), "Consistent");
|
||||
|
||||
// SpillSlots allocation is deferred.
|
||||
if (SpillSlots.empty()) {
|
||||
@@ -340,7 +348,7 @@ private:
|
||||
// Now that we've spilled the value, take it out of the register file
|
||||
FreeReg(Reg);
|
||||
AnySpilled = true;
|
||||
};
|
||||
}
|
||||
|
||||
void RemapReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
@@ -350,7 +358,7 @@ private:
|
||||
if (Index < SSAToReg.size()) {
|
||||
SSAToReg[Index] = Reg;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// Record a given assignment of register Reg to Node.
|
||||
void SetReg(Ref Node, PhysicalRegister Reg) {
|
||||
@@ -363,7 +371,7 @@ private:
|
||||
|
||||
RemapReg(Node, Reg);
|
||||
Node->Reg = Reg.Raw;
|
||||
};
|
||||
}
|
||||
|
||||
// Assign a register for a given Node, spilling if necessary.
|
||||
void AssignReg(IROp_Header* IROp, IROp_CodeBlock* Block, Ref CodeNode, IROp_Header* Pivot) {
|
||||
@@ -419,7 +427,7 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
RegClass ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegClass ClassType = GetRegClassFromNode(IROp);
|
||||
RegisterClassData* Class = &Classes[FEXCore::ToUnderlying(ClassType)];
|
||||
|
||||
// Spill to make room in the register file.
|
||||
@@ -432,7 +440,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(Class->Available != 0, "Post-condition of spilling");
|
||||
unsigned Reg = std::countr_zero(Class->Available);
|
||||
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount) {
|
||||
@@ -441,7 +449,7 @@ void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount)
|
||||
Classes[FEXCore::ToUnderlying(Class)].Count = RegisterCount;
|
||||
}
|
||||
|
||||
inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
|
||||
static bool KillMove(const IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
|
||||
// 32-bit moves in x86_64 are represented as a Bfe, detect them.
|
||||
if (LastOp->Op == OP_BFE && LastOp->C<IR::IROp_Bfe>()->lsb == 0 && LastOp->C<IR::IROp_Bfe>()->Width == 32) {
|
||||
auto Op = IROp->Op;
|
||||
@@ -459,7 +467,7 @@ inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref C
|
||||
return LastOp->Op == OP_STOREREGISTER;
|
||||
}
|
||||
|
||||
inline bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Size) {
|
||||
static bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Size) {
|
||||
if (IROp->Op == OP_SBFE) {
|
||||
auto Sbfe = IROp->C<IR::IROp_Sbfe>();
|
||||
return Sbfe->Width == 1 && Sbfe->lsb == (IR::OpSizeAsBits(Size) - 1) && Sbfe->Src == Src;
|
||||
@@ -468,7 +476,7 @@ inline bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Si
|
||||
}
|
||||
}
|
||||
|
||||
inline bool IsZero(const IROp_Header* IROp) {
|
||||
static bool IsZero(const IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT && IROp->C<IROp_Constant>()->Constant == 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,10 +6,11 @@ $end_info$
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
enum class RegClass : uint32_t;
|
||||
@@ -18,6 +19,9 @@ class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
public:
|
||||
virtual void AddRegisters(RegClass Class, uint32_t RegisterCount) = 0;
|
||||
|
||||
void SetNumPairRegs(uint32_t NumRegs);
|
||||
|
||||
protected:
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs {};
|
||||
};
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
@@ -32,14 +33,14 @@
|
||||
namespace FEXCore::IR {
|
||||
|
||||
// FIXME(pmatos): copy from OpcodeDispatcher.h
|
||||
inline uint32_t MMBaseOffset() {
|
||||
static uint32_t MMBaseOffset() {
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
|
||||
// Similar helper to the one in OpcodeDispatcher.h except we do not
|
||||
// need to handle flags, etc.
|
||||
template<typename T>
|
||||
void DeriveOp(Ref& RefV, IROps NewOp, IREmitter::IRPair<T> Expr) {
|
||||
static void DeriveOp(Ref& RefV, IROps NewOp, IREmitter::IRPair<T> Expr) {
|
||||
Expr.first->Header.Op = NewOp;
|
||||
RefV = Expr;
|
||||
}
|
||||
@@ -52,8 +53,8 @@ template<typename T>
|
||||
class FixedSizeStack {
|
||||
public:
|
||||
struct StackSlotEntry final {
|
||||
StackSlot Type;
|
||||
T Value;
|
||||
StackSlot Type = StackSlot::UNUSED;
|
||||
T Value = T::Invalid;
|
||||
};
|
||||
|
||||
static constexpr uint8_t size = 8;
|
||||
@@ -64,8 +65,7 @@ public:
|
||||
// If SlowPath is true, then TopOffset is always zero.
|
||||
int8_t TopOffset = 0;
|
||||
|
||||
FixedSizeStack()
|
||||
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T::Invalid}) {}
|
||||
FixedSizeStack() = default;
|
||||
|
||||
void push(const T& Value) {
|
||||
rotate();
|
||||
@@ -92,25 +92,23 @@ public:
|
||||
return buffer[Offset];
|
||||
}
|
||||
|
||||
void setTop(T Value, size_t Offset = 0) {
|
||||
void setTop(const T& Value, size_t Offset = 0) {
|
||||
buffer[Offset] = {StackSlot::VALID, Value};
|
||||
}
|
||||
|
||||
bool isValid(size_t Offset) const {
|
||||
return buffer[Offset].first;
|
||||
return buffer[Offset].Type == StackSlot::VALID;
|
||||
}
|
||||
|
||||
void clear() {
|
||||
for (auto& Elem : buffer) {
|
||||
Elem = {StackSlot::UNUSED, T::Invalid};
|
||||
}
|
||||
buffer.fill({StackSlot::UNUSED, T::Invalid});
|
||||
TopOffset = 0;
|
||||
}
|
||||
|
||||
void dump() const {
|
||||
LogMan::Msg::DFmt("-- Stack");
|
||||
|
||||
for (size_t i = 0; i < 8; i++) {
|
||||
for (size_t i = 0; i < buffer.size(); i++) {
|
||||
const auto& [Valid, Element] = buffer[i];
|
||||
if (Valid == StackSlot::VALID) {
|
||||
LogMan::Msg::DFmt("| ST{}: 0x{:x}", i, (uintptr_t)(Element.StackDataNode));
|
||||
@@ -126,7 +124,7 @@ public:
|
||||
}
|
||||
|
||||
// Returns a mask to set in AbridgedTagWord
|
||||
uint8_t getValidMask() {
|
||||
uint8_t getValidMask() const {
|
||||
uint8_t Mask = 0;
|
||||
for (size_t i = 0; i < buffer.size(); i++) {
|
||||
if (buffer[i].Type == StackSlot::VALID) {
|
||||
@@ -137,7 +135,7 @@ public:
|
||||
}
|
||||
|
||||
// Returns a mask to set in AbridgedTagWord
|
||||
uint8_t getInvalidMask() {
|
||||
uint8_t getInvalidMask() const {
|
||||
uint8_t Mask = 0;
|
||||
for (size_t i = 0; i < buffer.size(); i++) {
|
||||
if (buffer[i].Type == StackSlot::INVALID) {
|
||||
@@ -148,7 +146,7 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
fextl::vector<StackSlotEntry> buffer;
|
||||
std::array<StackSlotEntry, size> buffer {};
|
||||
};
|
||||
|
||||
class X87StackOptimization final : public Pass {
|
||||
@@ -201,11 +199,11 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
void StoreStackMem_Helper(const IRListView& IR, const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
LOGMAN_THROW_A_FMT(!ReducedPrecisionMode, "Full precision mode expected.");
|
||||
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
Ref AddrNode = IR.GetNode(Op->Addr);
|
||||
Ref Offset = IR.GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
@@ -229,11 +227,11 @@ private:
|
||||
|
||||
// Performs a store to memory from a value the stack passed in as StackNode.
|
||||
// This is the version dealing with the reduced precision case.
|
||||
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
void StoreStackMem_Reduced_Helper(const IRListView& IR, const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
LOGMAN_THROW_A_FMT(ReducedPrecisionMode, "Reduced precision mode expected.");
|
||||
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
Ref AddrNode = IR.GetNode(Op->Addr);
|
||||
Ref Offset = IR.GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
@@ -274,7 +272,7 @@ private:
|
||||
Ref LoadStackValueAtOffset_Slow(uint8_t Offset = 0);
|
||||
void StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset = 0, bool SetValid = true);
|
||||
// Update Top value in slow path for a pop
|
||||
void UpdateTopForPop_Slow();
|
||||
void UpdateTopForPop_Slow(bool InvalidateTag = true);
|
||||
void UpdateTopForPush_Slow();
|
||||
// Synchronizes the current simulated stack with the actual values.
|
||||
// Returns a new value for Top, that's synchronized between the simulated stack
|
||||
@@ -292,10 +290,10 @@ private:
|
||||
void Reset();
|
||||
|
||||
struct StackMemberInfo {
|
||||
StackMemberInfo() = delete;
|
||||
StackMemberInfo(Ref Data)
|
||||
constexpr StackMemberInfo() = default;
|
||||
constexpr StackMemberInfo(Ref Data)
|
||||
: StackDataNode(Data) {}
|
||||
StackMemberInfo(Ref Data, Ref Source, OpSize Size)
|
||||
constexpr StackMemberInfo(Ref Data, Ref Source, OpSize Size)
|
||||
: StackDataNode(Data)
|
||||
, Source({Size, Source}) {}
|
||||
Ref StackDataNode {}; // Reference to the data in the Stack.
|
||||
@@ -357,9 +355,9 @@ private:
|
||||
// On the slow path TopCache is always the last obtained version of top.
|
||||
// TopOffset is ignored
|
||||
bool SlowPath = false;
|
||||
|
||||
// Keeping IREmitter not to pass arguments around
|
||||
IREmitter* IREmit = nullptr;
|
||||
IRListView* IR = nullptr;
|
||||
};
|
||||
|
||||
inline const X87StackOptimization::StackMemberInfo X87StackOptimization::StackMemberInfo::Invalid {nullptr};
|
||||
@@ -575,25 +573,39 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
|
||||
HandleBinopValue(Op64, VFOp64, Op80, DestStackOffset, StackOffset2 != DestStackOffset, StackOffset1, StackNode, Reverse);
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::UpdateTopForPop_Slow() {
|
||||
inline void X87StackOptimization::UpdateTopForPop_Slow(bool InvalidateTag) {
|
||||
const auto PopContainer = [](auto& container) {
|
||||
const auto begin = std::begin(container);
|
||||
std::rotate(begin, std::next(begin), std::end(container));
|
||||
};
|
||||
|
||||
if (InvalidateTag) {
|
||||
SetX87ValidTag(0, false);
|
||||
}
|
||||
|
||||
// Pop the top of the x87 stack
|
||||
GetOffsetTopWithCache_Slow(1);
|
||||
std::rotate(TopOffsetCache.begin(), std::next(TopOffsetCache.begin()), TopOffsetCache.end());
|
||||
std::rotate(TopOffsetAddressCache.begin(), std::next(TopOffsetAddressCache.begin()), TopOffsetAddressCache.end());
|
||||
std::rotate(TopValueCache.begin(), std::next(TopValueCache.begin()), TopValueCache.end());
|
||||
std::rotate(FlushValuesPending.begin(), std::next(FlushValuesPending.begin()), FlushValuesPending.end());
|
||||
std::rotate(TopValidCache.begin(), std::next(TopValidCache.begin()), TopValidCache.end());
|
||||
PopContainer(TopOffsetCache);
|
||||
PopContainer(TopOffsetAddressCache);
|
||||
PopContainer(TopValueCache);
|
||||
PopContainer(FlushValuesPending);
|
||||
PopContainer(TopValidCache);
|
||||
FlushTopPending = true;
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::UpdateTopForPush_Slow() {
|
||||
// Pop the top of the x87 stack
|
||||
const auto PushContainer = [](auto& container) {
|
||||
const auto end = std::end(container);
|
||||
std::rotate(std::begin(container), std::prev(end), end);
|
||||
};
|
||||
|
||||
// Push the top of the x87 stack
|
||||
GetOffsetTopWithCache_Slow(1, true);
|
||||
std::rotate(TopOffsetCache.begin(), std::prev(TopOffsetCache.end()), TopOffsetCache.end());
|
||||
std::rotate(TopOffsetAddressCache.begin(), std::prev(TopOffsetAddressCache.end()), TopOffsetAddressCache.end());
|
||||
std::rotate(TopValueCache.begin(), std::prev(TopValueCache.end()), TopValueCache.end());
|
||||
std::rotate(FlushValuesPending.begin(), std::prev(FlushValuesPending.end()), FlushValuesPending.end());
|
||||
std::rotate(TopValidCache.begin(), std::prev(TopValidCache.end()), TopValidCache.end());
|
||||
PushContainer(TopOffsetCache);
|
||||
PushContainer(TopOffsetAddressCache);
|
||||
PushContainer(TopValueCache);
|
||||
PushContainer(FlushValuesPending);
|
||||
PushContainer(TopValidCache);
|
||||
FlushTopPending = true;
|
||||
}
|
||||
|
||||
@@ -724,7 +736,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
// Initialize IREmit member
|
||||
IREmit = Emit;
|
||||
IR = &CurrentIR;
|
||||
|
||||
// Run optimization proper
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
@@ -933,7 +944,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
UpdateTopForPush_Slow();
|
||||
StoreStackValueAtOffset_Slow(SourceNode);
|
||||
} else {
|
||||
auto* SourceNode = CurrentIR.GetNode(Op->X80Src);
|
||||
if (Op->OriginalValue.IsInvalid()) {
|
||||
// No original value to track - just push the converted data
|
||||
StackData.push(StackMemberInfo {SourceNode});
|
||||
@@ -1019,11 +1029,11 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
}
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
StoreStackMem_Reduced_Helper(Op, StackNode);
|
||||
StoreStackMem_Reduced_Helper(CurrentIR, Op, StackNode);
|
||||
break;
|
||||
}
|
||||
|
||||
StoreStackMem_Helper(Op, StackNode);
|
||||
StoreStackMem_Helper(CurrentIR, Op, StackNode);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1031,22 +1041,17 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
const auto* Op = IROp->C<IROp_StoreStackToStack>();
|
||||
auto Offset = Op->StackLocation;
|
||||
|
||||
if (Offset != 0) {
|
||||
auto Value = MigrateToSlowPath_IfInvalid();
|
||||
auto Value = MigrateToSlowPath_IfInvalid();
|
||||
|
||||
// Need to store st0 to stack location - basically a copy.
|
||||
if (SlowPath) {
|
||||
StoreStackValueAtOffset_Slow(LoadStackValueAtOffset_Slow(), Offset);
|
||||
} else {
|
||||
StackData.setTop(*Value, Offset);
|
||||
}
|
||||
// Need to store st0 to stack location - basically a copy.
|
||||
if (SlowPath) {
|
||||
StoreStackValueAtOffset_Slow(LoadStackValueAtOffset_Slow(), Offset);
|
||||
} else {
|
||||
StackData.setTop(*Value, Offset);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_POPSTACKDESTROY: {
|
||||
if (SlowPath) {
|
||||
SetX87ValidTag(0, false);
|
||||
}
|
||||
StackPop();
|
||||
break;
|
||||
}
|
||||
@@ -1067,8 +1072,8 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// Slow path: do actual memory operations
|
||||
Ref ValueTop = LoadStackValue();
|
||||
Ref ValueOffset = LoadStackValue(Offset);
|
||||
StoreStackValue(ValueOffset);
|
||||
StoreStackValue(ValueTop, Offset);
|
||||
StoreStackValue(ValueOffset, 0, true);
|
||||
StoreStackValue(ValueTop, Offset, true);
|
||||
} else {
|
||||
// Fast path: swap complete StackMemberInfo preserving Source metadata
|
||||
StackData.setTop(StackMemberOffset, 0);
|
||||
@@ -1087,7 +1092,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
ResultNode = IREmit->_VFNeg(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
|
||||
ResultNode = IREmit->_VXor(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
ResultNode = IREmit->_VXor(OpSize::i128Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -1102,7 +1107,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
} else {
|
||||
// Intermediate insts
|
||||
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
|
||||
ResultNode = IREmit->_VAndn(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
ResultNode = IREmit->_VAndn(OpSize::i128Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -1172,7 +1177,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_INCSTACKTOP: {
|
||||
if (SlowPath) {
|
||||
UpdateTopForPop_Slow();
|
||||
UpdateTopForPop_Slow(false);
|
||||
} else {
|
||||
StackData.rotate(false);
|
||||
}
|
||||
@@ -1227,8 +1232,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
SynchronizeStackValues();
|
||||
FlushCachedRegs();
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const HostFeatures& Features, OpSize GPROpSize) {
|
||||
|
||||
@@ -38,15 +38,13 @@ namespace FEXCore::Allocator {
|
||||
MMAP_Hook mmap {::mmap};
|
||||
MUNMAP_Hook munmap {::munmap};
|
||||
|
||||
uint64_t HostVASize {};
|
||||
|
||||
using GLIBC_MALLOC_Hook = void* (*)(size_t, const void* caller);
|
||||
using GLIBC_REALLOC_Hook = void* (*)(void*, size_t, const void* caller);
|
||||
using GLIBC_FREE_Hook = void (*)(void*, const void* caller);
|
||||
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Alloc64 {};
|
||||
static fextl::unique_ptr<Alloc::HostAllocator> Alloc64 {};
|
||||
|
||||
void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
static void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
void* Result = Alloc64->Mmap(addr, length, prot, flags, fd, offset);
|
||||
if (Result >= (void*)-4096) {
|
||||
errno = -(uint64_t)Result;
|
||||
@@ -70,7 +68,7 @@ void VirtualName(const char* Name, void* Ptr, size_t Size) {
|
||||
}
|
||||
}
|
||||
|
||||
int FEX_munmap(void* addr, size_t length) {
|
||||
static int FEX_munmap(void* addr, size_t length) {
|
||||
int Result = Alloc64->Munmap(addr, length);
|
||||
|
||||
if (Result != 0) {
|
||||
@@ -104,9 +102,11 @@ void ClearHooks() {
|
||||
}
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
|
||||
if (HostVASize) {
|
||||
return HostVASize;
|
||||
FEX_DEFAULT_VISIBILITY size_t GetHostVABits() {
|
||||
static uint64_t HostVABits = 0;
|
||||
|
||||
if (HostVABits) {
|
||||
return HostVABits;
|
||||
}
|
||||
|
||||
static constexpr std::array<uintptr_t, 7> TLBSizes = {
|
||||
@@ -125,6 +125,7 @@ FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
|
||||
::munmap(Ptr, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
if (Ptr != (void*)~0ULL || errno == EEXIST) {
|
||||
HostVABits = Bits;
|
||||
return Bits;
|
||||
}
|
||||
}
|
||||
@@ -217,7 +218,12 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(alloca(0));
|
||||
|
||||
const int MapsFD = open("/proc/self/maps", O_RDONLY);
|
||||
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
|
||||
if (MapsFD == -1) {
|
||||
// No procfs, as in BuildKit's emulator probe (an empty chroot). The assert
|
||||
// above it compiles out in release, and CollectMemoryGaps then spins on
|
||||
// read(-1) forever; reserve nothing instead.
|
||||
return {};
|
||||
}
|
||||
|
||||
auto Regions = CollectMemoryGaps(Begin, End, MapsFD);
|
||||
close(MapsFD);
|
||||
@@ -273,7 +279,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
}
|
||||
|
||||
fextl::vector<MemoryRegion> Setup48BitAllocatorIfExists(size_t PageSize) {
|
||||
size_t Bits = FEXCore::Allocator::DetermineVASize();
|
||||
size_t Bits = FEXCore::Allocator::GetHostVABits();
|
||||
if (Bits < 48) {
|
||||
return {};
|
||||
}
|
||||
@@ -281,6 +287,9 @@ fextl::vector<MemoryRegion> Setup48BitAllocatorIfExists(size_t PageSize) {
|
||||
uintptr_t Begin48BitVA = 0x0'8000'0000'0000ULL;
|
||||
uintptr_t End48BitVA = 0x1'0000'0000'0000ULL;
|
||||
auto Regions = StealMemoryRegion(Begin48BitVA, End48BitVA);
|
||||
if (Regions.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocatorWithRegions(Regions);
|
||||
AssignHookOverrides(PageSize);
|
||||
@@ -316,6 +325,7 @@ VirtualTHPPtr VirtualTHPControl {VirtualTHPNOP};
|
||||
void SetupHooks(size_t PageSize, HookPtrs Ptrs) {
|
||||
VirtualName = Ptrs.VirtualName;
|
||||
VirtualTHPControl = Ptrs.VirtualTHPControl;
|
||||
SetupAllocatorHooks(VirtualName);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -7,11 +7,9 @@
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -114,24 +112,24 @@ private:
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::SizeInBytes(NumElements);
|
||||
}
|
||||
|
||||
static void InitializeVMARegionUsed(LiveVMARegion* Region, size_t AdditionalSize) {
|
||||
size_t SizeOfLiveRegion =
|
||||
static void InitializeVMARegionUsed(LiveVMARegion* Region) {
|
||||
const size_t SizeOfLiveRegion =
|
||||
FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(Region->SlabInfo->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = SizeOfLiveRegion + AdditionalSize;
|
||||
|
||||
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
|
||||
Region->FreeSpace = Region->SlabInfo->RegionSize - SizeOfLiveRegion;
|
||||
|
||||
size_t NumManagedPages = SizePlusManagedData >> FEXCore::Utils::FEX_PAGE_SHIFT;
|
||||
size_t NumManagedPages = SizeOfLiveRegion >> FEXCore::Utils::FEX_PAGE_SHIFT;
|
||||
size_t ManagedSize = NumManagedPages << FEXCore::Utils::FEX_PAGE_SHIFT;
|
||||
|
||||
// Use madvise to set the full tracking region to zero.
|
||||
// This ensures unused pages are zero, while not having the backing pages consuming memory.
|
||||
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FEXCore::Utils::FEX_PAGE_SHIFT) - ManagedSize,
|
||||
MADV_DONTNEED);
|
||||
auto* MemoryAsBytes = reinterpret_cast<uint8_t*>(Region->UsedPages.Memory);
|
||||
const auto TrackingRegionSize = Region->SlabInfo->RegionSize - ManagedSize;
|
||||
::madvise(MemoryAsBytes + ManagedSize, TrackingRegionSize, MADV_DONTNEED);
|
||||
|
||||
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
|
||||
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
|
||||
::madvise(Region->UsedPages.Memory, ManagedSize, MADV_WILLNEED);
|
||||
::madvise(MemoryAsBytes, ManagedSize, MADV_WILLNEED);
|
||||
|
||||
// Set our reserved pages
|
||||
Region->UsedPages.MemSet(NumManagedPages);
|
||||
@@ -154,28 +152,27 @@ private:
|
||||
FEXCore::ForkableUniqueMutex AllocationMutex;
|
||||
void DetermineVASize();
|
||||
|
||||
LiveVMARegion* MakeRegionActive(ReservedRegionListType::iterator ReservedIterator, uint64_t UsedSize) {
|
||||
LiveVMARegion* MakeRegionActive(ReservedRegionListType::iterator ReservedIterator) {
|
||||
ReservedVMARegion* ReservedRegion = *ReservedIterator;
|
||||
|
||||
ReservedRegions->erase(ReservedIterator);
|
||||
|
||||
// mprotect the new region we've allocated
|
||||
size_t SizeOfLiveRegion =
|
||||
const size_t SizeOfLiveRegion =
|
||||
FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(ReservedRegion->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
|
||||
|
||||
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
|
||||
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizeOfLiveRegion, PROT_READ | PROT_WRITE);
|
||||
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
|
||||
strerror(errno));
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData);
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ReservedRegion->Base), SizeOfLiveRegion);
|
||||
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
|
||||
|
||||
// Copy over the reserved data
|
||||
LiveRange->SlabInfo = ReservedRegion;
|
||||
|
||||
// Initialize VMA
|
||||
LiveVMARegion::InitializeVMARegionUsed(LiveRange, UsedSize);
|
||||
LiveVMARegion::InitializeVMARegionUsed(LiveRange);
|
||||
|
||||
// Add to our active tracked ranges
|
||||
auto LiveIter = LiveRegions->emplace_back(LiveRange);
|
||||
@@ -187,7 +184,7 @@ private:
|
||||
};
|
||||
|
||||
void OSAllocator_64Bit::DetermineVASize() {
|
||||
size_t Bits = FEXCore::Allocator::DetermineVASize();
|
||||
size_t Bits = FEXCore::Allocator::GetHostVABits();
|
||||
uintptr_t Size = 1ULL << Bits;
|
||||
|
||||
UPPER_BOUND = Size;
|
||||
@@ -224,7 +221,7 @@ OSAllocator_64Bit::LiveVMARegion* OSAllocator_64Bit::FindLiveRegionForAddress(ui
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base && AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
LiveRegion = MakeRegionActive(it);
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -394,7 +391,7 @@ again:
|
||||
size_t lengthPlusManagedData = length + lengthOfLiveRegion;
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
if ((*it)->RegionSize >= lengthPlusManagedData) {
|
||||
MakeRegionActive(it, 0);
|
||||
MakeRegionActive(it);
|
||||
goto again;
|
||||
}
|
||||
}
|
||||
@@ -623,14 +620,14 @@ fextl::unique_ptr<T> make_alloc_unique(FEXCore::Allocator::MemoryRegion& Base, A
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocatorWithRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions) {
|
||||
// This is a bit tricky as we can't allocate memory safely except from the Regions provided. Otherwise we might overwrite memory pages we
|
||||
// don't own. Scan the memory regions and find the smallest one.
|
||||
FEXCore::Allocator::MemoryRegion& Smallest = Regions[0];
|
||||
for (auto& it : Regions) {
|
||||
if (it.Size <= Smallest.Size) {
|
||||
Smallest = it;
|
||||
FEXCore::Allocator::MemoryRegion* Smallest = &Regions[0];
|
||||
for (auto& Region : Regions) {
|
||||
if (Region.Size <= Smallest->Size) {
|
||||
Smallest = &Region;
|
||||
}
|
||||
}
|
||||
|
||||
return make_alloc_unique<OSAllocator_64Bit>(Smallest, Regions);
|
||||
return make_alloc_unique<OSAllocator_64Bit>(*Smallest, Regions);
|
||||
}
|
||||
|
||||
} // namespace Alloc::OSAllocator
|
||||
|
||||
@@ -39,10 +39,10 @@ struct FlexBitSet final {
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void MemClear(size_t Elements) {
|
||||
memset(Memory, 0, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
memset(Memory, 0, SizeInBytes(Elements));
|
||||
}
|
||||
void MemSet(size_t Elements) {
|
||||
memset(Memory, 0xFF, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
memset(Memory, 0xFF, SizeInBytes(Elements));
|
||||
}
|
||||
|
||||
// Range scanning results
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
|
||||
#ifdef ENABLE_FEX_ALLOCATOR
|
||||
#include <rpmalloc/rpmalloc.h>
|
||||
#ifndef _WIN32
|
||||
@@ -20,16 +22,32 @@
|
||||
namespace FEXCore::Allocator {
|
||||
using mmap_hook_type = void* (*)(void* addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
using munmap_hook_type = int (*)(void* addr, size_t length);
|
||||
using vma_name_hook_type = void (*)(const char* name, const void* address, size_t size);
|
||||
|
||||
#ifdef ENABLE_FEX_ALLOCATOR
|
||||
typedef void* (*rp_mmap_hook_type)(size_t size, size_t alignment, size_t* offset, size_t* mapped_size);
|
||||
typedef void (*rp_munmap_hook_type)(void* address, size_t offset, size_t mapped_size);
|
||||
typedef void (*vma_name_hook_type)(const char* name, const void* address, size_t size);
|
||||
extern "C" rp_mmap_hook_type rp_mmap_hook;
|
||||
extern "C" rp_munmap_hook_type rp_munmap_hook;
|
||||
extern "C" vma_name_hook_type rp_name_hook;
|
||||
|
||||
#ifndef _WIN32
|
||||
mmap_hook_type fex_mmap_hook = ::mmap;
|
||||
munmap_hook_type fex_munmap_hook = ::munmap;
|
||||
|
||||
static inline void LocalVirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
#ifndef _WIN32
|
||||
static bool Supports {true};
|
||||
if (Supports) {
|
||||
auto Result = prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result == -1) {
|
||||
// Disable any additional attempts.
|
||||
Supports = false;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
// Assume a 64KB page size until told otherwise.
|
||||
@@ -172,6 +190,11 @@ void InitializeAllocator(size_t PageSize) {
|
||||
rpmalloc_initialize_config(&global_interface, &global_config);
|
||||
rp_mmap_hook = FEX_rp_mmap;
|
||||
rp_munmap_hook = FEX_rp_memory_unmap;
|
||||
rp_name_hook = LocalVirtualName;
|
||||
}
|
||||
#else
|
||||
void SetupAllocatorHooks(vma_name_hook_type NameHook) {
|
||||
rp_name_hook = NameHook;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
@@ -0,0 +1,366 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <dirent.h>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <windows.h>
|
||||
#include <winnt.h>
|
||||
#include <winternl.h>
|
||||
#include <ntstatus.h>
|
||||
|
||||
// FEX today doesn't have Windows include in FEXCore.
|
||||
extern "C" NTSTATUS WINAPI NtQueryDirectoryFile(HANDLE, HANDLE, PIO_APC_ROUTINE, PVOID, PIO_STATUS_BLOCK, PVOID, ULONG,
|
||||
FILE_INFORMATION_CLASS, BOOLEAN, PUNICODE_STRING, BOOLEAN);
|
||||
|
||||
extern "C" NTSTATUS RtlUnicodeToUTF8N(OUT PCHAR UTF8StringDestination, IN ULONG UTF8StringMaxByteCount,
|
||||
OUT PULONG UTF8StringActualByteCount, IN PCWCH UnicodeStringSource, IN ULONG UnicodeStringByteCount);
|
||||
#endif
|
||||
|
||||
namespace FEXCore::FileUtils {
|
||||
#ifndef _WIN32
|
||||
static inline bool unlinkat(int fd, const char* path, bool dir) {
|
||||
if (::unlinkat(fd, path, dir ? AT_REMOVEDIR : 0) == -1) {
|
||||
return errno == ENOENT;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool RecursiveRemoveDirectory(int parent_fd, const char* Directory) {
|
||||
// Don't follow symlinks and ensure it closes on exec.
|
||||
constexpr int DIR_FLAGS = O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC;
|
||||
int dir_fd = ::openat(parent_fd, Directory, DIR_FLAGS);
|
||||
|
||||
if (dir_fd == -1) {
|
||||
if (errno == ENOENT) {
|
||||
// Probably raced something. Non-error.
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Walk the directory listing.
|
||||
bool Result = true;
|
||||
// Four pages arbitrary chosen to be a balance between NFS wanting to return data in page-size granules and
|
||||
// local filesystems returning arbitrary sizes.
|
||||
size_t dirent_size = 4096 * 4;
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
ssize_t read = getdents64(dir_fd, dirent_buffer, dirent_size);
|
||||
|
||||
if (read == -1) {
|
||||
if (errno == EINVAL) {
|
||||
// Buffer too small? Scale and try again.
|
||||
dirent_size *= 2;
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Anything else just exit.
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
if (read == 0) {
|
||||
// Done.
|
||||
break;
|
||||
}
|
||||
|
||||
for (size_t dirent_offset = 0; dirent_offset < read;) {
|
||||
auto path_dirent = reinterpret_cast<const struct dirent*>(dirent_buffer + dirent_offset);
|
||||
std::string_view path_name_view = path_dirent->d_name;
|
||||
|
||||
if (path_name_view == "." || path_name_view == "..") {
|
||||
// Skip these two special files.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (path_dirent->d_type == DT_DIR) {
|
||||
// Recurse directories as we find them and remove them.
|
||||
if (!RecursiveRemoveDirectory(dir_fd, path_dirent->d_name)) {
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
}
|
||||
|
||||
// Remove anything possible.
|
||||
if (!unlinkat(dir_fd, path_dirent->d_name, path_dirent->d_type == DT_DIR)) {
|
||||
// Couldn't unlink the file for some reason.
|
||||
LogMan::Msg::IFmt("Failed to remove file: {}", path_name_view);
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
// dirent is a VLA so we need to increment by reported size.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
}
|
||||
}
|
||||
|
||||
end:
|
||||
if (dirent_buffer) {
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
}
|
||||
|
||||
close(dir_fd);
|
||||
return Result;
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool RecursiveRemoveDirectory(const fextl::string& Directory) {
|
||||
return RecursiveRemoveDirectory(AT_FDCWD, Directory.c_str()) && unlinkat(AT_FDCWD, Directory.c_str(), true);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void WalkDirectory(std::string_view Directory,
|
||||
fextl::move_only_function<void(std::string_view name, bool is_dir, const void* user_data)> Callback,
|
||||
const void* user_data) {
|
||||
constexpr int DIR_FLAGS = O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC;
|
||||
int dir_fd = ::openat(AT_FDCWD, fextl::string(Directory).c_str(), DIR_FLAGS);
|
||||
if (dir_fd == -1) {
|
||||
if (errno == ENOENT) {
|
||||
// Probably raced something. Non-error.
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Four pages arbitrary chosen to be a balance between NFS wanting to return data in page-size granules and
|
||||
// local filesystems returning arbitrary sizes.
|
||||
size_t dirent_size = 4096 * 4;
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
ssize_t read = getdents64(dir_fd, dirent_buffer, dirent_size);
|
||||
|
||||
if (read == -1) {
|
||||
if (errno == EINVAL) {
|
||||
// Buffer too small? Scale and try again.
|
||||
dirent_size *= 2;
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Anything else just exit.
|
||||
goto end;
|
||||
}
|
||||
|
||||
if (read == 0) {
|
||||
// Done.
|
||||
break;
|
||||
}
|
||||
|
||||
for (size_t dirent_offset = 0; dirent_offset < read;) {
|
||||
auto path_dirent = reinterpret_cast<const struct dirent*>(dirent_buffer + dirent_offset);
|
||||
std::string_view path_name_view = path_dirent->d_name;
|
||||
|
||||
if (path_name_view == "." || path_name_view == "..") {
|
||||
// Skip these two special files.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
continue;
|
||||
}
|
||||
|
||||
Callback(path_name_view, path_dirent->d_type == DT_DIR, user_data);
|
||||
|
||||
// dirent is a VLA so we need to increment by reported size.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
}
|
||||
}
|
||||
|
||||
end:
|
||||
if (dirent_buffer) {
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
}
|
||||
|
||||
close(dir_fd);
|
||||
}
|
||||
|
||||
#else
|
||||
static inline std::optional<UNICODE_STRING> PathToNTPath(std::string_view Path) {
|
||||
UNICODE_STRING PathW;
|
||||
if (!RtlCreateUnicodeStringFromAsciiz(&PathW, fextl::string(Path).c_str())) {
|
||||
return std::nullopt;
|
||||
}
|
||||
UNICODE_STRING NTPath;
|
||||
bool Success = RtlDosPathNameToNtPathName_U(PathW.Buffer, &NTPath, nullptr, nullptr);
|
||||
RtlFreeUnicodeString(&PathW);
|
||||
if (!Success) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
return NTPath;
|
||||
}
|
||||
|
||||
static inline void FreeNTPath(UNICODE_STRING Path) {
|
||||
RtlFreeUnicodeString(&Path);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void WalkDirectory(std::string_view Directory,
|
||||
fextl::move_only_function<void(std::string_view name, bool is_dir, const void* user_data)> Callback,
|
||||
const void* user_data) {
|
||||
auto NTPath = PathToNTPath(Directory);
|
||||
if (!NTPath) {
|
||||
return;
|
||||
}
|
||||
|
||||
NTSTATUS Status {};
|
||||
|
||||
HANDLE dir_fd {};
|
||||
OBJECT_ATTRIBUTES attr {};
|
||||
IO_STATUS_BLOCK io {};
|
||||
|
||||
InitializeObjectAttributes(&attr, &*NTPath, OBJ_CASE_INSENSITIVE, nullptr, nullptr);
|
||||
Status = NtOpenFile(&dir_fd, FILE_LIST_DIRECTORY | SYNCHRONIZE, &attr, &io, FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE,
|
||||
FILE_DIRECTORY_FILE | FILE_SYNCHRONOUS_IO_NONALERT);
|
||||
|
||||
FreeNTPath(*NTPath);
|
||||
|
||||
if (!NT_SUCCESS(Status)) {
|
||||
return;
|
||||
}
|
||||
|
||||
BOOLEAN FirstQuery = TRUE;
|
||||
size_t dirent_size = 4096 * 4;
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
Status = NtQueryDirectoryFile(dir_fd,
|
||||
nullptr, // Event
|
||||
nullptr, // ApcRoutine
|
||||
nullptr, // ApcContext
|
||||
&io, dirent_buffer, dirent_size, FileDirectoryInformation,
|
||||
FALSE, // ReturnSingleEntry
|
||||
nullptr, // FileName
|
||||
FirstQuery);
|
||||
FirstQuery = FALSE;
|
||||
|
||||
if (Status == STATUS_NO_MORE_FILES) {
|
||||
// No more files
|
||||
break;
|
||||
}
|
||||
|
||||
if (!NT_SUCCESS(Status)) {
|
||||
if (Status == STATUS_BUFFER_TOO_SMALL || Status == STATUS_INFO_LENGTH_MISMATCH) {
|
||||
// Buffer too small? Scale and try again.
|
||||
dirent_size *= 2;
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Any other failure, exit loop.
|
||||
break;
|
||||
}
|
||||
|
||||
// Iterate the entries returned.
|
||||
for (size_t dirent_offset = 0;;) {
|
||||
auto Info = reinterpret_cast<FILE_DIRECTORY_INFORMATION*>(dirent_buffer + dirent_offset);
|
||||
|
||||
std::wstring_view EntryName(Info->FileName, Info->FileNameLength / sizeof(wchar_t));
|
||||
|
||||
if (EntryName != L"." && EntryName != L"..") {
|
||||
ULONG utf8_bytes_needed = 0;
|
||||
ULONG actual_utf8_bytes = 0;
|
||||
RtlUnicodeToUTF8N(nullptr, 0, &utf8_bytes_needed, Info->FileName, Info->FileNameLength);
|
||||
char* dynamic_buf = reinterpret_cast<char*>(FEXCore::Allocator::malloc(utf8_bytes_needed + 1));
|
||||
if (dynamic_buf) {
|
||||
RtlUnicodeToUTF8N(dynamic_buf, utf8_bytes_needed + 1, &actual_utf8_bytes, Info->FileName, Info->FileNameLength);
|
||||
std::string_view name_view(dynamic_buf, actual_utf8_bytes);
|
||||
bool is_dir = (Info->FileAttributes & FILE_ATTRIBUTE_DIRECTORY) != 0;
|
||||
Callback(name_view, is_dir, user_data);
|
||||
FEXCore::Allocator::free(dynamic_buf);
|
||||
}
|
||||
}
|
||||
|
||||
if (Info->NextEntryOffset == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
dirent_offset += Info->NextEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
end:
|
||||
if (dirent_buffer) {
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
}
|
||||
|
||||
NtClose(dir_fd);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool RecursiveRemoveDirectory(const fextl::string& Directory) {
|
||||
using CallbackType = void (*)(std::string_view name, bool is_dir, const void* user_data);
|
||||
struct UserData {
|
||||
CallbackType RemoveFile {};
|
||||
std::string_view base_path;
|
||||
};
|
||||
|
||||
CallbackType RemoveFile = [](std::string_view name, bool is_dir, const void* user_data) {
|
||||
auto Data = reinterpret_cast<const UserData*>(user_data);
|
||||
|
||||
auto full_path = std::format("{}/{}", Data->base_path, name);
|
||||
if (is_dir) {
|
||||
UserData NewData {
|
||||
.RemoveFile = Data->RemoveFile,
|
||||
.base_path = full_path,
|
||||
};
|
||||
|
||||
WalkDirectory(full_path, Data->RemoveFile, &NewData);
|
||||
}
|
||||
|
||||
if (is_dir) {
|
||||
RemoveDirectoryA(full_path.c_str());
|
||||
} else {
|
||||
DeleteFileA(full_path.c_str());
|
||||
}
|
||||
};
|
||||
|
||||
UserData Data {
|
||||
.RemoveFile = RemoveFile,
|
||||
.base_path = Directory,
|
||||
};
|
||||
|
||||
WalkDirectory(Directory, RemoveFile, &Data);
|
||||
|
||||
return RemoveDirectoryA(Directory.c_str()) != 0;
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::FileUtils
|
||||
@@ -1,4 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Assert {
|
||||
// This function can not be inlined
|
||||
[[noreturn]]
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
|
||||
namespace FEXCore::UncheckedLongJump {
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#ifdef __arm64ec__
|
||||
#pragma clang diagnostic ignored "-Winline-asm"
|
||||
#endif
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
__asm volatile(R"(
|
||||
|
||||
@@ -7,7 +7,8 @@
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Threads {
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread> CreateThread_Default(ThreadFunc Func, void* Arg) {
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread>
|
||||
CreateThread_Default(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName) {
|
||||
ERROR_AND_DIE_FMT("Frontend didn't setup thread creation!");
|
||||
}
|
||||
|
||||
@@ -20,8 +21,9 @@ static FEXCore::Threads::Pointers Ptrs = {
|
||||
.CleanupAfterFork = CleanupAfterFork_Default,
|
||||
};
|
||||
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> FEXCore::Threads::Thread::Create(ThreadFunc Func, void* Arg) {
|
||||
return Ptrs.CreateThread(Func, Arg);
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread>
|
||||
FEXCore::Threads::Thread::Create(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName) {
|
||||
return Ptrs.CreateThread(Func, Arg, Flags, ThreadName);
|
||||
}
|
||||
|
||||
void FEXCore::Threads::Thread::CleanupAfterFork() {
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/WorkQueueThread.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
WorkQueueThread::WorkQueueThread(FEXCore::Threads::Flags ThreadFlags, const char* ThreadName) {
|
||||
Thread = FEXCore::Threads::Thread::Create(ThreadEntry, this, ThreadFlags, ThreadName);
|
||||
}
|
||||
|
||||
WorkQueueThread::~WorkQueueThread() {
|
||||
{
|
||||
std::unique_lock lk {Mutex};
|
||||
Stop = true;
|
||||
}
|
||||
CV.notify_one();
|
||||
|
||||
if (Thread && Thread->joinable()) {
|
||||
Thread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void WorkQueueThread::QueueWork(fextl::unique_ptr<WorkItem> Work) {
|
||||
{
|
||||
std::unique_lock lk {Mutex};
|
||||
Queue.push_back(std::move(Work));
|
||||
}
|
||||
CV.notify_one();
|
||||
}
|
||||
|
||||
void WorkQueueThread::ThreadProc() {
|
||||
while (true) {
|
||||
fextl::unique_ptr<WorkItem> Work;
|
||||
{
|
||||
std::unique_lock lk {Mutex};
|
||||
while (!(Stop || !Queue.empty())) {
|
||||
CV.wait(lk);
|
||||
}
|
||||
if (Queue.empty()) {
|
||||
// nothing to do? must be stopping
|
||||
LOGMAN_THROW_A_FMT(Stop, "WorkQueueThread wakes up empty but no Stop?");
|
||||
return;
|
||||
}
|
||||
|
||||
Work = std::move(Queue.front());
|
||||
Queue.pop_front();
|
||||
}
|
||||
|
||||
Work->Run();
|
||||
// Work is destroyed here
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -0,0 +1,584 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <bit>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
/**
|
||||
* A bitset that supports allocating contiguous ranges atomically.
|
||||
* - Lock-free, with the caveat that on contention for > 64-bit it would be faster to acquire a lock.
|
||||
* - If low-contention then atomic-behaviour wins.
|
||||
* - Three modes of allocation:
|
||||
* - 1-bit, single-atomic.
|
||||
* - <= 64-bit, single-atomic, contained within single word (introduces sparsity).
|
||||
* - > 64-bit, multiple-atomic, contiguous, roll-back on contiguous allocation failure.
|
||||
* - Can return failure to allocate even if there is space in certain circumstances.
|
||||
* - If the allocated size crosses multiple words.
|
||||
* - Race to allocation caused contention.
|
||||
* - Remembers last allocation/free for inner-word allocations to improve performance.
|
||||
* - Large greater than atomic-word scans always scan from the start.
|
||||
* - Resetting the bitset with clear() is lower cost when page_size=true.
|
||||
* - MADV_DONTNEED replaces pages with zero-page
|
||||
* - When page_size=false, basic memset is also fairly quick.
|
||||
*/
|
||||
template<bool track_last_allocation = false, bool page_sized = true>
|
||||
class atomic_bitset final {
|
||||
public:
|
||||
void init(void* ptr, size_t bits) {
|
||||
LOGMAN_THROW_A_FMT(bits != 0, "Can't init zero");
|
||||
|
||||
base = reinterpret_cast<uint64_t*>(ptr);
|
||||
bits_to_track = bits;
|
||||
words_to_track = bits / WORD_SIZE_BITS;
|
||||
LOGMAN_THROW_A_FMT(bits % WORD_SIZE_BITS == 0, "Bits to track must match uint64_t");
|
||||
if constexpr (page_sized) {
|
||||
LOGMAN_THROW_A_FMT(bits % (4096 * 8) == 0, "Bits to track must match bit count in page");
|
||||
}
|
||||
|
||||
last_allocation_track.set_last_allocation(0);
|
||||
}
|
||||
|
||||
// Allocate a contiguous buffer of bits.
|
||||
// Returns initial bit offset on success, ~0ULL on failure.
|
||||
size_t allocate(size_t count) {
|
||||
LOGMAN_THROW_A_FMT(count != 0, "Can't allocate zero");
|
||||
LOGMAN_THROW_A_FMT(count <= bits_to_track, "Can't allocate larger than size");
|
||||
|
||||
if (count == 1) [[likely]] {
|
||||
// Common and trivial case.
|
||||
return allocate_one(last_allocation_track.get_last_allocation(), words_to_track);
|
||||
} else if (count <= WORD_SIZE_BITS) [[likely]] {
|
||||
// Allocate up to a single word. Don't allow cross-word allocations
|
||||
// Could cause some sparsity
|
||||
return allocate_inside_word(count, last_allocation_track.get_last_allocation(), words_to_track);
|
||||
}
|
||||
|
||||
// TODO: Always scans from beginning to end.
|
||||
// Support iterative scanning.
|
||||
return allocate_large_amount(count, 0, words_to_track);
|
||||
}
|
||||
|
||||
// Frees a contiguous set of bits.
|
||||
void free(size_t index, size_t count) {
|
||||
LOGMAN_THROW_A_FMT(count != 0, "Can't free zero");
|
||||
LOGMAN_THROW_A_FMT(index < bits_to_track, "Can't free beyond end");
|
||||
|
||||
if (count == 1) [[likely]] {
|
||||
free_one(index);
|
||||
return;
|
||||
} else if ((index % WORD_SIZE_BITS + count) <= WORD_SIZE_BITS) [[likely]] {
|
||||
free_inside_word(index, count);
|
||||
return;
|
||||
}
|
||||
|
||||
free_large_amount(index, count);
|
||||
}
|
||||
|
||||
// Clears the entire bitset.
|
||||
// Not thread safe!
|
||||
void clear() {
|
||||
const size_t bytes = words_to_track * sizeof(uint64_t);
|
||||
if constexpr (page_sized) {
|
||||
// VirtualDontNeed replaces pages with zero page.
|
||||
FEXCore::Allocator::VirtualDontNeed(base, bytes);
|
||||
} else {
|
||||
memset(base, 0, bytes);
|
||||
}
|
||||
|
||||
last_allocation_track.set_last_allocation(0);
|
||||
}
|
||||
|
||||
// Checks if a single bit is set.
|
||||
bool is_set(size_t index) const {
|
||||
const size_t word_index = index / WORD_SIZE_BITS;
|
||||
const size_t word_offset = index % WORD_SIZE_BITS;
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
|
||||
const uint64_t bit_mask = 1ULL << word_offset;
|
||||
return (word_atomic.load() & bit_mask) != 0;
|
||||
}
|
||||
|
||||
size_t size_in_bits() const {
|
||||
return bits_to_track;
|
||||
}
|
||||
|
||||
constexpr static size_t invalid() {
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
// non-atomically returns the number of set bits in the bitset.
|
||||
size_t popcount() const {
|
||||
size_t count {};
|
||||
|
||||
// Just ensure all store are visible.
|
||||
std::atomic_thread_fence(std::memory_order_release);
|
||||
|
||||
for (size_t word_index = 0; word_index < words_to_track; ++word_index) {
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
count += std::popcount(word_atomic.load(std::memory_order_relaxed));
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
private:
|
||||
uint64_t* base {};
|
||||
size_t bits_to_track {};
|
||||
size_t words_to_track {};
|
||||
struct data_to_track_nop {
|
||||
constexpr static size_t get_last_allocation() {
|
||||
return 0;
|
||||
}
|
||||
constexpr static void set_last_allocation(size_t) {}
|
||||
};
|
||||
|
||||
struct data_to_track {
|
||||
std::atomic<size_t> last_allocation_word {};
|
||||
|
||||
constexpr size_t get_last_allocation() const {
|
||||
// It's okay if this isn't up to date, full scan of the region still occurs.
|
||||
return last_allocation_word.load(std::memory_order_relaxed);
|
||||
}
|
||||
constexpr void set_last_allocation(size_t word) {
|
||||
last_allocation_word = word;
|
||||
}
|
||||
};
|
||||
using data_type = typename std::conditional<track_last_allocation, data_to_track, data_to_track_nop>::type;
|
||||
data_type last_allocation_track {};
|
||||
|
||||
constexpr static size_t WORD_SIZE_BITS = sizeof(uint64_t) * 8;
|
||||
|
||||
size_t allocate_one(size_t beginning_word_index, size_t ending_word_index) {
|
||||
// Trivial spin.
|
||||
for (size_t i = beginning_word_index; i < ending_word_index; ++i) {
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[i]);
|
||||
auto expected_word = word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
if (expected_word == ~0ULL) {
|
||||
// Won't pass.
|
||||
continue;
|
||||
}
|
||||
|
||||
// Spin on the word trying to acquire a bit.
|
||||
// Uncontended case should immediately succeed.
|
||||
// Contended case can spin the whole word and lose every acquire.
|
||||
|
||||
do {
|
||||
const auto zero_bit = std::countr_one(expected_word);
|
||||
const auto bit_mask = 1ULL << zero_bit;
|
||||
|
||||
// If mask was already set, then we raced to acquire (returned value will be 1).
|
||||
// If mask not set, then we will have acquired (returned value will be 0).
|
||||
expected_word = word_atomic.fetch_or(bit_mask);
|
||||
if ((expected_word & bit_mask) == 0) {
|
||||
// Acquired the bit, return the offset.
|
||||
last_allocation_track.set_last_allocation(i);
|
||||
return i * WORD_SIZE_BITS + zero_bit;
|
||||
}
|
||||
|
||||
// Bit was already acquired.
|
||||
expected_word |= bit_mask;
|
||||
} while (expected_word != ~0ULL);
|
||||
}
|
||||
|
||||
if constexpr (track_last_allocation) {
|
||||
if (beginning_word_index) {
|
||||
// One more chance to get an allocation.
|
||||
// Scan before the previous allocation to see if any free slots have appeared.
|
||||
return allocate_one(0, beginning_word_index);
|
||||
}
|
||||
}
|
||||
|
||||
// Failure to acquire here.
|
||||
return invalid();
|
||||
}
|
||||
|
||||
size_t allocate_inside_word(size_t count, size_t beginning_word_index, size_t ending_word_index) {
|
||||
for (size_t i = beginning_word_index; i < ending_word_index; ++i) {
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[i]);
|
||||
auto expected_word = word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
// Spin on the word trying to acquire a bit.
|
||||
// Uncontended case should immediately succeed.
|
||||
// Contended case can spin the whole word and lose every acquire.
|
||||
while (expected_word != ~0ULL) {
|
||||
auto zero_bit = std::countr_one(expected_word);
|
||||
bool fits = false;
|
||||
uint64_t bit_mask = count == WORD_SIZE_BITS ? ~0ULL : ((1ULL << count) - 1);
|
||||
for (; (zero_bit + count) <= WORD_SIZE_BITS; ++zero_bit) {
|
||||
// Check if the bits fit.
|
||||
uint64_t tmp_bit_mask = bit_mask << zero_bit;
|
||||
if ((expected_word & tmp_bit_mask) == 0) {
|
||||
fits = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Could never fit, early exit.
|
||||
if (!fits) {
|
||||
break;
|
||||
}
|
||||
|
||||
// Shift bit_mask to the desired location.
|
||||
bit_mask <<= zero_bit;
|
||||
|
||||
uint64_t desired_word {};
|
||||
bool acquired = true;
|
||||
|
||||
do {
|
||||
if (expected_word & bit_mask) {
|
||||
// Couldn't acquire this field, move to the next.
|
||||
acquired = false;
|
||||
break;
|
||||
}
|
||||
|
||||
// We desire setting a single bit.
|
||||
desired_word = expected_word | bit_mask;
|
||||
} while (!word_atomic.compare_exchange_strong(expected_word, desired_word));
|
||||
|
||||
if (acquired) {
|
||||
// Acquired the bit, return the offset.
|
||||
last_allocation_track.set_last_allocation(i);
|
||||
return i * WORD_SIZE_BITS + zero_bit;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if constexpr (track_last_allocation) {
|
||||
if (beginning_word_index) {
|
||||
// One more chance to get an allocation.
|
||||
// Scan before the previous allocation to see if any free slots have appeared.
|
||||
return allocate_inside_word(count, 0, beginning_word_index);
|
||||
}
|
||||
}
|
||||
|
||||
// Failure to acquire here.
|
||||
return invalid();
|
||||
}
|
||||
|
||||
size_t allocate_large_amount(size_t count, size_t beginning_word_index, size_t ending_word_index) {
|
||||
// This version of the code needs to explicitly deal with large allocations that can't fit in a word.
|
||||
// So we are always scanning minimum 2 words.
|
||||
const size_t num_words_to_scan = FEXCore::AlignUpPowerOf2(count, WORD_SIZE_BITS) / WORD_SIZE_BITS;
|
||||
const size_t last_word_to_scan = ending_word_index - num_words_to_scan - 1;
|
||||
|
||||
// Scan forward to find the first word.
|
||||
for (size_t base_index = beginning_word_index; base_index < last_word_to_scan;) {
|
||||
uint64_t leading_zeros {};
|
||||
|
||||
size_t center_word_index = 1;
|
||||
bool has_center {};
|
||||
|
||||
size_t tail_bits {};
|
||||
|
||||
size_t remaining_bits = count;
|
||||
|
||||
auto check_head_fitment = [&]() -> bool {
|
||||
auto base_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
auto base_expected_word = base_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
// Count the leading zeros, if it is above zero then we can start here.
|
||||
leading_zeros = std::countl_zero(base_expected_word);
|
||||
|
||||
if (leading_zeros == 0) {
|
||||
// Nope.
|
||||
++base_index;
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto check_center_fitment = [&]() -> bool {
|
||||
// Subtract the number of zeros.
|
||||
remaining_bits -= leading_zeros;
|
||||
|
||||
has_center = remaining_bits >= WORD_SIZE_BITS;
|
||||
|
||||
while (remaining_bits >= WORD_SIZE_BITS) {
|
||||
// All words in-between head and tail must be zero.
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[base_index + center_word_index]);
|
||||
auto center_expected_word = center_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
if (center_expected_word != 0) {
|
||||
// Couldn't fit, won't ever fit in this range, so jump ahead to the last scanned item.
|
||||
// Might still be able to start on the tail of this word.
|
||||
base_index += center_word_index;
|
||||
return false;
|
||||
}
|
||||
|
||||
++center_word_index;
|
||||
remaining_bits -= WORD_SIZE_BITS;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto check_tail_fitment = [&]() -> bool {
|
||||
tail_bits = remaining_bits;
|
||||
if (tail_bits) {
|
||||
// Now for the tail (if it is necessary).
|
||||
size_t tail_index = center_word_index;
|
||||
|
||||
// Count the trailing zeros, if it fits out remaining bits then we can try and allocate.
|
||||
auto tail_word_atomic = std::atomic_ref<uint64_t>(base[base_index + tail_index]);
|
||||
auto tail_expected_word = tail_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
const auto trailing_zeros = std::countr_zero(tail_expected_word);
|
||||
|
||||
if (trailing_zeros < remaining_bits) {
|
||||
// Couldn't fit, but also won't ever fit in this range. Jump ahead to this tail item.
|
||||
// Might still be able to start on the tail of this word.
|
||||
base_index += tail_index;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
if (!check_head_fitment()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!check_center_fitment()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!check_tail_fitment()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// We can try fitting!
|
||||
if (attempt_allocate_range_from_base(base_index, count, leading_zeros, tail_bits, has_center)) {
|
||||
const size_t head_leading_offset = (WORD_SIZE_BITS - leading_zeros);
|
||||
return base_index * WORD_SIZE_BITS + head_leading_offset;
|
||||
}
|
||||
|
||||
// Failure to fit here means we could never fit in this full range. Jump past all the bits
|
||||
// Rescanning the tail to avoid fragmented sparsity on the tails.
|
||||
base_index += num_words_to_scan - 1;
|
||||
}
|
||||
|
||||
return invalid();
|
||||
}
|
||||
|
||||
bool attempt_allocate_range_from_base(size_t base_index, size_t count, size_t head_bits, size_t tail_bits, bool has_center) {
|
||||
const bool has_tail = tail_bits != 0;
|
||||
|
||||
const uint64_t head_bit_mask = head_bits == WORD_SIZE_BITS ? ~0ULL : (((1ULL << head_bits) - 1) << (WORD_SIZE_BITS - head_bits));
|
||||
const uint64_t tail_bit_mask = (1ULL << tail_bits) - 1;
|
||||
const uint64_t center_words_count = (count - head_bits - tail_bits) / WORD_SIZE_BITS;
|
||||
const uint64_t center_words_base_index = base_index + 1;
|
||||
const uint64_t tail_word_base_index = center_words_base_index + center_words_count;
|
||||
|
||||
// Three distinct sections, all of which need to support rewinding.
|
||||
// - Head: Setting the leading zeros to one
|
||||
// - Always exists. Can be a full word, or partial.
|
||||
// - Center: Setting all in-between words to ~0ULL
|
||||
// - Might not exist if tail is smaller than a word
|
||||
// - Always full words if it does exist.
|
||||
// - Tail: Set all trailing zeros up to the size to 1
|
||||
// - Might not exist if Center perfectly aligned to word edge.
|
||||
// - Always partial words, otherwise it would be considered "Center".
|
||||
|
||||
bool set_head {true};
|
||||
bool set_center {true};
|
||||
bool set_tail {true};
|
||||
size_t num_center_set {};
|
||||
|
||||
// Head first.
|
||||
auto set_head_word = [&]() -> bool {
|
||||
auto head_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
auto head_expected_word = head_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
uint64_t desired_word {};
|
||||
do {
|
||||
if (head_expected_word & head_bit_mask) {
|
||||
// Another thread raced and allocated.
|
||||
return false;
|
||||
}
|
||||
|
||||
// Set the whole mask in one atomic operation.
|
||||
desired_word = head_expected_word | head_bit_mask;
|
||||
} while (!head_word_atomic.compare_exchange_strong(head_expected_word, desired_word));
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto clear_head_word = [&]() {
|
||||
auto head_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
head_word_atomic.fetch_and(~head_bit_mask);
|
||||
};
|
||||
|
||||
auto set_center_words = [&]() -> bool {
|
||||
const size_t end_center_word_index = center_words_base_index + center_words_count;
|
||||
for (size_t center_index = center_words_base_index; center_index < end_center_word_index; ++center_index) {
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[center_index]);
|
||||
auto center_expected_word = center_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
// Center words set a full ~0ULL mask.
|
||||
const uint64_t desired_word {~0ULL};
|
||||
do {
|
||||
if (center_expected_word) {
|
||||
// Another thread raced and allocated.
|
||||
return false;
|
||||
}
|
||||
} while (!center_word_atomic.compare_exchange_strong(center_expected_word, desired_word));
|
||||
|
||||
++num_center_set;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto clear_center_words = [&]() {
|
||||
const size_t end_center_word_index = center_words_base_index + num_center_set;
|
||||
for (size_t center_index = center_words_base_index; center_index < end_center_word_index; ++center_index) {
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[center_index]);
|
||||
center_word_atomic.store(0);
|
||||
}
|
||||
};
|
||||
|
||||
auto set_tail_word = [&]() -> bool {
|
||||
auto tail_word_atomic = std::atomic_ref<uint64_t>(base[tail_word_base_index]);
|
||||
auto tail_expected_word = tail_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
uint64_t desired_word {};
|
||||
do {
|
||||
if (tail_expected_word & tail_bit_mask) {
|
||||
// Another thread raced and allocated.
|
||||
return false;
|
||||
}
|
||||
|
||||
// Set the whole mask in one atomic operation.
|
||||
desired_word = tail_expected_word | tail_bit_mask;
|
||||
} while (!tail_word_atomic.compare_exchange_strong(tail_expected_word, desired_word));
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
set_head = set_head_word();
|
||||
|
||||
// Do the center if it exists.
|
||||
if (set_head && has_center) {
|
||||
set_center = set_center_words();
|
||||
}
|
||||
|
||||
// Do the tail if it exists.
|
||||
if (set_head && set_center && has_tail) {
|
||||
set_tail = set_tail_word();
|
||||
}
|
||||
|
||||
if (set_head && set_center && set_tail) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Some stage failed to set, rewind everything.
|
||||
if (set_head) {
|
||||
// Clear the head if it was set
|
||||
clear_head_word();
|
||||
}
|
||||
|
||||
if (has_center && num_center_set) {
|
||||
// `set_center` might not be set, but it still managed to set some of the words.
|
||||
clear_center_words();
|
||||
}
|
||||
|
||||
// Tail doesn't need to rewind as it will never have been set if we got here.
|
||||
return false;
|
||||
}
|
||||
|
||||
#if defined(__ARM_FEATURE_ATOMICS) && __ARM_FEATURE_ATOMICS == 1
|
||||
// Might violate memory-ordering requirements?
|
||||
// TODO: Verify and enable or delete depending.
|
||||
// Provides an 11% (Cortex-X4) to 25% (AmpereOneA) performance improvement.
|
||||
constexpr static bool use_stclr {};
|
||||
static inline void stclr(uint64_t value, uint64_t* addr) {
|
||||
asm volatile("stclrl %[Val], [%[addr]];" ::[Val] "r"(value), [addr] "r"(addr) : "memory");
|
||||
}
|
||||
#endif
|
||||
|
||||
void free_one(size_t index) {
|
||||
const size_t word_index = index / WORD_SIZE_BITS;
|
||||
const size_t word_offset = index % WORD_SIZE_BITS;
|
||||
last_allocation_track.set_last_allocation(word_index);
|
||||
|
||||
const uint64_t bic_bit_mask = 1ULL << word_offset;
|
||||
#if defined(__ARM_FEATURE_ATOMICS) && __ARM_FEATURE_ATOMICS == 1
|
||||
if constexpr (use_stclr) {
|
||||
stclr(bic_bit_mask, &base[word_index]);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
word_atomic.fetch_and(~bic_bit_mask);
|
||||
}
|
||||
|
||||
void free_inside_word(size_t index, size_t count) {
|
||||
const size_t word_index = index / WORD_SIZE_BITS;
|
||||
const size_t word_offset = index % WORD_SIZE_BITS;
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
last_allocation_track.set_last_allocation(word_index);
|
||||
|
||||
if (count == WORD_SIZE_BITS) {
|
||||
word_atomic.store(0);
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t bic_bit_mask = ((1ULL << count) - 1) << word_offset;
|
||||
#if defined(__ARM_FEATURE_ATOMICS) && __ARM_FEATURE_ATOMICS == 1
|
||||
if constexpr (use_stclr) {
|
||||
stclr(bic_bit_mask, &base[word_index]);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
word_atomic.fetch_and(~bic_bit_mask);
|
||||
}
|
||||
|
||||
void free_large_amount(size_t index, size_t count) {
|
||||
// Incoming count can be less than WORD_SIZE_BITS if it is unaligned and crossing multiple words.
|
||||
// Needs to always handle at minimum a head plus center and/or tail arrangement.
|
||||
const size_t base_index = index / WORD_SIZE_BITS;
|
||||
const size_t last_index = FEXCore::AlignUpPowerOf2(index + count, WORD_SIZE_BITS) / WORD_SIZE_BITS;
|
||||
const size_t num_words_to_scan = last_index - base_index;
|
||||
LOGMAN_THROW_A_FMT(num_words_to_scan > 1, "Needs to be larger than 1 ({}, {})", index, count);
|
||||
|
||||
size_t remaining_bits = count;
|
||||
|
||||
const uint64_t head_offset_start = index % WORD_SIZE_BITS;
|
||||
const uint64_t head_bits = WORD_SIZE_BITS - head_offset_start;
|
||||
|
||||
uint64_t head_mask = head_bits == WORD_SIZE_BITS ? ~0ULL : (((1ULL << head_bits) - 1) << head_offset_start);
|
||||
|
||||
remaining_bits -= head_bits;
|
||||
auto head_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
head_word_atomic.fetch_and(~head_mask);
|
||||
|
||||
const size_t remaining_center_words = remaining_bits / WORD_SIZE_BITS;
|
||||
for (size_t i = 0; i < remaining_center_words; ++i) {
|
||||
// Handle centers if they exist.
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[base_index + i + 1]);
|
||||
center_word_atomic.store(0);
|
||||
remaining_bits -= WORD_SIZE_BITS;
|
||||
}
|
||||
|
||||
if (remaining_bits) {
|
||||
// Handle tail if they exist, must always be less than WORD_SIZE_BITS.
|
||||
LOGMAN_THROW_A_FMT(remaining_bits < WORD_SIZE_BITS, "Too large");
|
||||
const uint64_t tail_mask = (1ULL << remaining_bits) - 1;
|
||||
|
||||
auto tail_word_atomic = std::atomic_ref<uint64_t>(base[base_index + remaining_center_words + 1]);
|
||||
tail_word_atomic.fetch_and(~tail_mask);
|
||||
}
|
||||
}
|
||||
};
|
||||
} // namespace FEXCore::Utils
|
||||
Loaded 100 of 442 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user