mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 10:00:16 +02:00
Compare commits
260
Commits
FEX-2608
...
FEX-2609.1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9fbdc00bd6 | ||
|
|
116053cd29 | ||
|
|
536722f59d | ||
|
|
395b132f34 | ||
|
|
566a5e266d | ||
|
|
cc5686de37 | ||
|
|
049c932675 | ||
|
|
84fabf4ed8 | ||
|
|
310c2504f6 | ||
|
|
021c4fa4bf | ||
|
|
319c5bf023 | ||
|
|
023d0cdaa6 | ||
|
|
5a4f84cad9 | ||
|
|
eb27391954 | ||
|
|
ffaa5d6316 | ||
|
|
8bf22b6382 | ||
|
|
42c663269b | ||
|
|
983c5defc1 | ||
|
|
c92805475d | ||
|
|
50e6eee95a | ||
|
|
31cda95039 | ||
|
|
970ce4bc13 | ||
|
|
e4c6154be2 | ||
|
|
064a6e96c7 | ||
|
|
d704bd3cef | ||
|
|
9f511d5d78 | ||
|
|
a77d00acae | ||
|
|
5d11b8b0d5 | ||
|
|
bf70dee3ae | ||
|
|
4a065ad1a2 | ||
|
|
96d955e7b1 | ||
|
|
abec314720 | ||
|
|
ec63be4028 | ||
|
|
179b4423a7 | ||
|
|
9956a528b9 | ||
|
|
e3c6b37f70 | ||
|
|
b78bde7ec8 | ||
|
|
6ff7e82e3b | ||
|
|
37675e0351 | ||
|
|
38c9f14dbd | ||
|
|
7187bc0e59 | ||
|
|
c5700cb0cc | ||
|
|
8ed4c52d58 | ||
|
|
bd65c3e568 | ||
|
|
7052f91495 | ||
|
|
040d3115d2 | ||
|
|
7236c2bf0a | ||
|
|
ed3fa2f1c6 | ||
|
|
a454845ffd | ||
|
|
40bfff985f | ||
|
|
96d65cd6bc | ||
|
|
e4e45bcc59 | ||
|
|
45fbab212a | ||
|
|
8243b93e4d | ||
|
|
52cc5689db | ||
|
|
cfef3d2cbf | ||
|
|
fd64ea54fe | ||
|
|
d1b2333195 | ||
|
|
97fbd11177 | ||
|
|
e7836f26dd | ||
|
|
cf6da6ac26 | ||
|
|
a0cb91daa4 | ||
|
|
e010e8adf0 | ||
|
|
7fe7c39a00 | ||
|
|
b03dc847f5 | ||
|
|
ac6be77592 | ||
|
|
35330df728 | ||
|
|
bd9897b164 | ||
|
|
e704275ffd | ||
|
|
e33cb3c831 | ||
|
|
91264f05e2 | ||
|
|
30e4650a8b | ||
|
|
add92f681b | ||
|
|
21b8f8e20d | ||
|
|
ca055cd7d5 | ||
|
|
42e207932b | ||
|
|
3157685f0f | ||
|
|
d833f3ae67 | ||
|
|
fbf5187794 | ||
|
|
f2c8adf7a3 | ||
|
|
d472ce190c | ||
|
|
c4c82f23ee | ||
|
|
20e9e2044e | ||
|
|
1a0412969b | ||
|
|
55c80e6c4f | ||
|
|
83469208c9 | ||
|
|
3471ca67c4 | ||
|
|
4048faa54b | ||
|
|
61d94ac117 | ||
|
|
511c45c4c6 | ||
|
|
bce8b13366 | ||
|
|
4a7fda4516 | ||
|
|
b29561e131 | ||
|
|
5a68359f37 | ||
|
|
00b1249e53 | ||
|
|
8be7c1bf8e | ||
|
|
aaca8772bd | ||
|
|
ab6e67596f | ||
|
|
433091c326 | ||
|
|
c01bfcd52f | ||
|
|
673a387808 | ||
|
|
a6e74cdbe6 | ||
|
|
ff4da8a620 | ||
|
|
1558a2c65d | ||
|
|
66cad978c3 | ||
|
|
c11a0cef68 | ||
|
|
8cf2bf7ad8 | ||
|
|
ef5439e5ae | ||
|
|
c0b24b2f2e | ||
|
|
0d3a5f96f0 | ||
|
|
73dc3b3edd | ||
|
|
01bc89b82e | ||
|
|
2f663db52c | ||
|
|
e17fdf9e90 | ||
|
|
b695c5ae11 | ||
|
|
f5935b4006 | ||
|
|
eca6569bd3 | ||
|
|
51c241d1f8 | ||
|
|
83ead40b78 | ||
|
|
dea5c62a44 | ||
|
|
b3f902166b | ||
|
|
98964c5527 | ||
|
|
8ccc8dab49 | ||
|
|
e1d8881f57 | ||
|
|
9742a560f5 | ||
|
|
7a8b9dd341 | ||
|
|
063af711af | ||
|
|
aed71ed5e9 | ||
|
|
8b9237fd52 | ||
|
|
fd180a16d3 | ||
|
|
2a82f58195 | ||
|
|
34ffd2d6e2 | ||
|
|
62c9c130c4 | ||
|
|
fca23d88c3 | ||
|
|
3a179a3a40 | ||
|
|
fe17c447d9 | ||
|
|
b21a8dafa7 | ||
|
|
926a871dd4 | ||
|
|
79ef5031eb | ||
|
|
f0135eb332 | ||
|
|
566b25a35b | ||
|
|
e6748ea1ed | ||
|
|
5ab2c723ff | ||
|
|
cac8785aca | ||
|
|
0814780c2f | ||
|
|
753bce5e73 | ||
|
|
d04ad79eaf | ||
|
|
baca86bc3f | ||
|
|
adfff26d58 | ||
|
|
ef3a6963f3 | ||
|
|
19dba4a3ec | ||
|
|
49c1cda5c0 | ||
|
|
2ea749c7ea | ||
|
|
9d4b493ec0 | ||
|
|
7e3e4f81ac | ||
|
|
e740af05b0 | ||
|
|
4ed80fd071 | ||
|
|
59116a06b9 | ||
|
|
a409220e54 | ||
|
|
485f1c6acf | ||
|
|
6646a5cc72 | ||
|
|
2a76d3b153 | ||
|
|
7614e48322 | ||
|
|
631b8c3f58 | ||
|
|
b4b38a92ef | ||
|
|
c3769a6937 | ||
|
|
8d69852785 | ||
|
|
b83dd97762 | ||
|
|
22498dc871 | ||
|
|
96172840f1 | ||
|
|
e8880e4e98 | ||
|
|
26f6cda589 | ||
|
|
f0b45010ec | ||
|
|
ad3939a44f | ||
|
|
c9b23eb0e7 | ||
|
|
44f69f4dd7 | ||
|
|
c3fb6ccaaa | ||
|
|
b47a36a47b | ||
|
|
af9b438eec | ||
|
|
fa9bbf081b | ||
|
|
db3817a260 | ||
|
|
635befb4c8 | ||
|
|
a69daa2524 | ||
|
|
b563703701 | ||
|
|
0717f4689b | ||
|
|
bf30f6af4c | ||
|
|
dfcbdac347 | ||
|
|
8d12f3d5aa | ||
|
|
02f8ab5f87 | ||
|
|
b2b6b263bb | ||
|
|
e3f208f61f | ||
|
|
008d990f20 | ||
|
|
c038bb5794 | ||
|
|
efc5eaf9c7 | ||
|
|
c1e9d19809 | ||
|
|
85efb47a3b | ||
|
|
bc69693b6c | ||
|
|
c9c5a75b76 | ||
|
|
f50279a2e7 | ||
|
|
561c32b45d | ||
|
|
f42dc71972 | ||
|
|
ea67665bbf | ||
|
|
c3b4d4b7bb | ||
|
|
492dac719f | ||
|
|
9377bac5e7 | ||
|
|
e0bbfdda84 | ||
|
|
9618b5adef | ||
|
|
a3d609ebb1 | ||
|
|
2b0d94536f | ||
|
|
73ab3bc56d | ||
|
|
412c49a8ce | ||
|
|
6f29dfcbb8 | ||
|
|
f3ab82a73f | ||
|
|
6734c9ed3e | ||
|
|
71afe47675 | ||
|
|
7075377a63 | ||
|
|
7c1036df09 | ||
|
|
a312347589 | ||
|
|
f386c62dba | ||
|
|
30a81484d7 | ||
|
|
00a6b046a8 | ||
|
|
c156498c5c | ||
|
|
adea3e410f | ||
|
|
40940ae0b3 | ||
|
|
1bab28dad4 | ||
|
|
4838265589 | ||
|
|
f6d20a1a88 | ||
|
|
11be444d45 | ||
|
|
363bf85b5b | ||
|
|
720039ca6d | ||
|
|
4e9399e029 | ||
|
|
466ee3cbcf | ||
|
|
390d5d2fb4 | ||
|
|
b543df8299 | ||
|
|
de16c18961 | ||
|
|
2158309c51 | ||
|
|
430846d7f2 | ||
|
|
f4e362c477 | ||
|
|
7966bfb077 | ||
|
|
a4e04dd7b4 | ||
|
|
b161a74365 | ||
|
|
fd141ed6d7 | ||
|
|
e2ff0dd1cd | ||
|
|
6284c39eb3 | ||
|
|
7f0bdf8d63 | ||
|
|
015beff4cc | ||
|
|
92e43c25d4 | ||
|
|
69fe85274f | ||
|
|
0122ef9e83 | ||
|
|
c772c0e4e7 | ||
|
|
c171c06192 | ||
|
|
9365e6240b | ||
|
|
f8e3571c66 | ||
|
|
6ebcf65451 | ||
|
|
4844729e93 | ||
|
|
9686454161 | ||
|
|
167d79f0fa | ||
|
|
b1275edb63 | ||
|
|
681636cd68 | ||
|
|
34a87cc94c |
No files matched your search
+21
-8
@@ -428,11 +428,16 @@ else ()
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
if (MINGW)
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
else()
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
find_package(range-v3 QUIET)
|
||||
@@ -480,12 +485,20 @@ endif()
|
||||
|
||||
set(FEX_TUNE_COMPILE_FLAGS)
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
set(TUNE_ARCH_STRING "${TUNE_ARCH}")
|
||||
if(ARCHITECTURE_arm64)
|
||||
set(TUNE_ARCH_STRING "${TUNE_ARCH}+crc")
|
||||
endif()
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH_STRING}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH_STRING}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH_STRING}' but the compiler doesn't support this")
|
||||
endif()
|
||||
elseif(ARCHITECTURE_arm64)
|
||||
# Need to always append crc
|
||||
check_cxx_compiler_flag("-march=armv8-a+crc" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=armv8-a+crc")
|
||||
endif()
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
|
||||
@@ -270,6 +270,9 @@ public:
|
||||
void fcvtxnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFloatConvertOdd(0b00, 0b10, pg, zn, zd);
|
||||
}
|
||||
void bfcvtnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFloatConvertOdd(0b10, 0b10, pg, zn, zd);
|
||||
}
|
||||
///< Size is destination size
|
||||
void fcvtnt(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i32Bit || size == SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
@@ -292,8 +295,6 @@ public:
|
||||
SVEFloatConvertOdd(ConvertedSrcSize, ConvertedDestSize, pg, zn, zd);
|
||||
}
|
||||
|
||||
// XXX: BFCVTNT
|
||||
|
||||
// SVE2 floating-point pairwise operations
|
||||
void faddp(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn, ZRegister zm) {
|
||||
SVEFloatPairwiseArithmetic(0b000, size, pg, zd, zn, zm);
|
||||
@@ -2312,15 +2313,15 @@ public:
|
||||
|
||||
// SVE floating-point convert precision
|
||||
void fcvt(SubRegSize to, SubRegSize from, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && from != SubRegSize::i8Bit, "Can't use 8-bit element size");
|
||||
SVEFPConvertPrecision(to, from, zd, pg, zn);
|
||||
}
|
||||
void fcvtx(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
uint32_t Instr = 0b0110'0101'0000'1010'1010'0000'0000'0000;
|
||||
Instr |= pg.Idx() << 10;
|
||||
Instr |= zn.Idx() << 5;
|
||||
Instr |= zd.Idx();
|
||||
dc32(Instr);
|
||||
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i8Bit, zd, pg, zn);
|
||||
}
|
||||
void bfcvt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i32Bit, zd, pg, zn);
|
||||
}
|
||||
|
||||
// SVE floating-point unary operations
|
||||
@@ -3847,14 +3848,19 @@ private:
|
||||
|
||||
void SVEFPConvertPrecision(SubRegSize to, SubRegSize from, ZRegister zd, PRegister pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && to != SubRegSize::i128Bit && from != SubRegSize::i8Bit && from != SubRegSize::i128Bit,
|
||||
"Can't use 8-bit or 128-bit element size");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i128Bit && from != SubRegSize::i128Bit, "Can't use 128-bit element size");
|
||||
|
||||
// Encodings for the to and from sizes can get a little funky
|
||||
// depending on what is being converted to/from.
|
||||
const uint32_t op = [&] {
|
||||
switch (from) {
|
||||
case SubRegSize::i8Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i32Bit: return 0x00020000U;
|
||||
default: return UINT32_MAX;
|
||||
}
|
||||
}
|
||||
|
||||
case SubRegSize::i16Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i32Bit: return 0x00810000U;
|
||||
@@ -3866,6 +3872,7 @@ private:
|
||||
case SubRegSize::i32Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i16Bit: return 0x00800000U;
|
||||
case SubRegSize::i32Bit: return 0x00820000U;
|
||||
case SubRegSize::i64Bit: return 0x00C30000U;
|
||||
default: return UINT32_MAX;
|
||||
}
|
||||
|
||||
@@ -9,8 +9,8 @@ set(CMAKE_AR ${MINGW_TRIPLE}-ar)
|
||||
# Compile everything as static to avoid requiring the MinGW runtime libraries, force page aligned sections so that
|
||||
# debug symbols work correctly, and disable loop alignment to workaround an LLVM bug
|
||||
# (https://github.com/llvm/llvm-project/issues/47432)
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_C_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
set(CMAKE_CXX_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
set(CMAKE_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
|
||||
+47
-47
@@ -210,53 +210,53 @@ click==8.1.7 \
|
||||
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
|
||||
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
|
||||
# via black
|
||||
cryptography==49.0.0 \
|
||||
--hash=sha256:026ac7423e6fa66872d3bf889be5974507da3944f866f704fa200eadacd00001 \
|
||||
--hash=sha256:07cab27cc7b7e0fd28e5e26bb9eeedde5c135c868b46de4a27845abe94af6122 \
|
||||
--hash=sha256:084ef1af862eb07ec46d25f68689f2102a9fc0e05ce7b80f14f5fe51e4eef0f6 \
|
||||
--hash=sha256:0b82e28ee398a386f0807bba7884d30f25218855690f45115831bcce5d90822c \
|
||||
--hash=sha256:0e959b578856a3924bc0cbb710fc12c387b9412a951389f3ca61704a9e25f325 \
|
||||
--hash=sha256:0f21641cf4b30fca7aee061ced0ec7ad7b073518088b7c9969a297c0ae796c69 \
|
||||
--hash=sha256:196ecd6a36e4e9aa10270393bb98d8df88fccee0bf1e5128b91ae4eb4375896d \
|
||||
--hash=sha256:2400ef9c9e2299a25614eb1dea3db54a69b1349efd043bfac9c67630d136df36 \
|
||||
--hash=sha256:28d8b15e6275f12c8a207dc309dfa957903c927d08d0cc937ee3f63f200693cc \
|
||||
--hash=sha256:2afe9051da7ae7bd5905da5a949280c7d2bb75682e188f650a9d0f2756b834c6 \
|
||||
--hash=sha256:2eda353d8a27bcbcaa4cbed18994a74ab4d19a2ca897db188ea269ab9b71419b \
|
||||
--hash=sha256:32703d93296f5c1f4b53349ad3a250c2cae0fdecd3a3dd5d47e616d8d616af27 \
|
||||
--hash=sha256:33cd0565932807baddb67b96dbee92f2c374b5c89dee09fd74079aeb8c8dba61 \
|
||||
--hash=sha256:35b151772baff2c74cba7fa290ceaff4c3b11c0c881eb93eb5dbc05a7cfbba18 \
|
||||
--hash=sha256:36d1709f992593689b45bda411498d62c6e365f2ca00b84657d4dadd24de16db \
|
||||
--hash=sha256:42b0684e0e40cf26122427802486f6d93aea593612603a94fbf260c7eb1e9c1b \
|
||||
--hash=sha256:4ae387c9cb68ea569ca17e490d66d8142b81c3cc814bf179974b7d146e490bbb \
|
||||
--hash=sha256:53ecee2e23f7169b6117e99fc8a944e5e50f79e69758a83b52a00cb98ab2b2d2 \
|
||||
--hash=sha256:66ec79c3904820572d7e987abdf304281f141d37ad9a489b8e97066e7b9b6459 \
|
||||
--hash=sha256:67e1d20ad9ef3a563c59ef22e7a8a0b8210bd26604369ea4a30a7c66aefe504e \
|
||||
--hash=sha256:6f2debedf9ca60cf1d5bd466475638af5130f89965605cd818484d19987d3a21 \
|
||||
--hash=sha256:6fc361c34fb6aac015ce19435876635e5c6d21db31998b0920f675f131e043b8 \
|
||||
--hash=sha256:73a205dce83953d131a4aa1e0fd917a2fd1c5b1eef251e9d7152efefcbf5caf7 \
|
||||
--hash=sha256:7abcee80084cda3f7691f3eb1ce480d8df49cec637b429aa35986c1de71738aa \
|
||||
--hash=sha256:8c25ceb16df5b9435f3f6a9829204985b0e0cbee3b48aacd432c7d2c850b44d9 \
|
||||
--hash=sha256:966fe0e9c67490071f14c0d2b1cb2dfb3023c5ce39457343931415f08382f2db \
|
||||
--hash=sha256:9e82dcc8e56052715fb18b2429e3bca4823b1629136a2084fc45a9a5cecb9b64 \
|
||||
--hash=sha256:b20133d204d2bb56ba047642199603876c872026ca53e79c35b83772ab2cc505 \
|
||||
--hash=sha256:b39efa323140595abd3ecca8529d321ae50f55f3aa3ba9cc81ea56a6011953d5 \
|
||||
--hash=sha256:b47db11c2c3525083296069b98ac5221907455e989ae0c2e3008bde851921615 \
|
||||
--hash=sha256:b87e65d263b3e5d3bb92a57e2a6638e2f31110fa7aa890c7b2dbba42248d0a3f \
|
||||
--hash=sha256:b970c6da94d5bb18629db453d14f2a1300f6bf59b61e9b82377931ef95504866 \
|
||||
--hash=sha256:be9fcb48a55f023493482827d4f459bd263cc20efde64f204b97c123201850c6 \
|
||||
--hash=sha256:c2bc30226390d60ea19d9f82b19db005fe0452154a23c1c410c12ea801e43561 \
|
||||
--hash=sha256:c83782480a4a9da4d0feb51950131ba32e12e70813848b3343f6e18c28a66838 \
|
||||
--hash=sha256:cbc77da8c523d5abd028635ba850a6966fcee2c82e2bf65a41d1d8afe0f98be9 \
|
||||
--hash=sha256:ccac2bfebc306b862133e3bb71f3f6ee8bb525240089b2d952e4144b3a6d5da7 \
|
||||
--hash=sha256:d0527ce944105f257f605a827d6ebead966c752038b6e8656abb9c5edee6fc68 \
|
||||
--hash=sha256:d8ecde755e2e91bf773fc94e8c9d730cd7f2007004cb492263a794ec3899a1c8 \
|
||||
--hash=sha256:e3fb64c420688e5319ae25113a354015abbd8dffbfbc41781a1ea66fc7622ac3 \
|
||||
--hash=sha256:e5dfc1e64de5677cec922ffa8da89c546d0415bf6efdf081842e5d44c84e1f0e \
|
||||
--hash=sha256:ec5e529fb80935c94fe7b729f9972b50e351a0e6b50aa294fd5cabb109fcc29a \
|
||||
--hash=sha256:f37d847238971164fdbc68ade6f6574aecc9c0af714190e2083429ff68f4ce9d \
|
||||
--hash=sha256:f78ff2c9ed8dc2d036b0f4d640e22522213d047c1b14e61205a7e55c80a494d4 \
|
||||
--hash=sha256:f89660a348f4f78a92366240a61404e337586ef7f5909a2fef59ca88ef505493 \
|
||||
--hash=sha256:fc1e275c2f1d97b1a6450b8b0ea3ebfa6e087a611c2b26cb2404d48588abab7b
|
||||
cryptography==50.0.0 \
|
||||
--hash=sha256:031e2d5dd4bb9caa3ca9c82e5a197fd8ae680232cee62603d1a813f3f07e3d03 \
|
||||
--hash=sha256:06a32a980526a6ab9a4b9bf8f7385800791e2bb960903cb6b530e4817509a3b7 \
|
||||
--hash=sha256:07479a1cb08219ab719147e742e76090c9c773321959bb94946fffdd397a6437 \
|
||||
--hash=sha256:07949c449a1abcf60d1ee6e88956d89404c7df3c8258f46589e912988e551987 \
|
||||
--hash=sha256:105110f43a471dbd0060b9c9516cb8a6a79233631a04cc2ba16f28323ac6e025 \
|
||||
--hash=sha256:11b74db56cdbe3cdee6e3f6982ecb70334fa10dce99ed58bf7894aaaa3b2a037 \
|
||||
--hash=sha256:12b9c6996425c76ea6c457ace4f3073e715b8c545add07cd1a8f3a4f90691269 \
|
||||
--hash=sha256:1489e263a8048bb8b6a8bac662eb2d402ea5d2b7b4699b72f385f1e2772db105 \
|
||||
--hash=sha256:19736989797678c6af1e55cd49055cdbcb55d8f6b5583ac5335f933aba9101dc \
|
||||
--hash=sha256:1b4a266766514614f8aa60416e71f2fc6e575d36e7bdc90f644fadb2f4b75b95 \
|
||||
--hash=sha256:2a8183b489dc1f7f80f135780fadc1108f14b31b8a40411c7a5b17425f65f28b \
|
||||
--hash=sha256:37fdb0d0111f1e2ff07139dfb79f1b49531f8e213c46f1163dd7642979b58c47 \
|
||||
--hash=sha256:3f5735ffe4996d28b809371756219f5354864902a3b9e7c0b9ee87041209fc9c \
|
||||
--hash=sha256:49e7d93abdbd2990caced757e5fade25302f719c3c8fb6e6fff2dde98999fc41 \
|
||||
--hash=sha256:5e34edd123674534acd70147f0ca331eaa2c74e6325fb2028c886aa26ba0b68c \
|
||||
--hash=sha256:62598a8a57f815db4c6259a4e97d857dab56697e7de8e8ab02352ab74da1995d \
|
||||
--hash=sha256:65c2c3add92b45fd0709db8594536aea39c2a67af0e27ffcf049c498501140b7 \
|
||||
--hash=sha256:6ba6a53445bd3cfa809ef3ef5f1589aa6ba08784a1d962bf47d0940e871dab1c \
|
||||
--hash=sha256:6e7d61120573a7f2cd94cc095f9e81f6967c61ccdf194285aa143ecec8e0b708 \
|
||||
--hash=sha256:7cec5b856506da6defb290f30c9ee687d5f5e8cb0bd3f6459dde43b0b4fa40ef \
|
||||
--hash=sha256:80b63928fa35083b33966ce1efb70e5b9607181e49dcd1c22c8c005e319f667f \
|
||||
--hash=sha256:82148ec5bddac30b51a5b3c1945075f896fa022cb93f8e4a01e9f6ee95292c5f \
|
||||
--hash=sha256:828743d939e9629bc267b8e2d08d8bb67cd4319c771a33d4b18b22dd8fb7440a \
|
||||
--hash=sha256:8d89f3976b10b4ce31118de72329025f70d2c6ead14a8217c5514dd2c6d5a78f \
|
||||
--hash=sha256:8eb5e1172eb569ea8a872796576e6a67c276351728b6455d5beb01242b027c6a \
|
||||
--hash=sha256:900131fafd8aead39ac7dd3a7e833be754c17a95cfd91221636949fe4eb0aa8a \
|
||||
--hash=sha256:910d11e1a385c654bf738bf3e6b8e6ed5de0f5610fcae2be9e5b398d8081d20e \
|
||||
--hash=sha256:910e1d2668e7de9648f2bcee30e180db2a6b15c30f887d7c4c93ddf96e3992e3 \
|
||||
--hash=sha256:9aa87839c383bdbab6ef865787a1fb877af8dd03464c4400322726feaaadfc6d \
|
||||
--hash=sha256:a1b30560f2acc95aa8b2e06e716a13dbfc97314747b80d9707e307f77b40d6b3 \
|
||||
--hash=sha256:a91296cb61e8df6f86d0c19cc4068228da256bf59bf86049fbd821084565327f \
|
||||
--hash=sha256:b42a28c1844fd9de8f3f7d540e36b66f3a9c83fceac7170ebc7a6a19edd9dcae \
|
||||
--hash=sha256:bd1c592e4d5974f0d08d4888e432157adba757c66da0246918e43677fafa2d30 \
|
||||
--hash=sha256:c87f62a3d3b9888ed0fdde100ec06aa61ca9cd44bad9057d1dff9a516b5f5bb9 \
|
||||
--hash=sha256:c99c003e088647b8a5b7c145d6f78c335f6348332b62e142d411c4b63d1460b9 \
|
||||
--hash=sha256:ccdc4a71a4dabae05de219404f9f4abc38e3b58422177ff93d0da05967dafa07 \
|
||||
--hash=sha256:d24fead1d4d076e1bfb006dcec392074a3cd8d7b4fc8a595aa64073b2b7a96ba \
|
||||
--hash=sha256:d58c3db7cd6eed54e6c06744db55456b65ebd7492ddeae9c1e93cfca7aa857d3 \
|
||||
--hash=sha256:d764dcf130c428ef66786f866dd750f53182bc608813489915e9fc106bb0c82f \
|
||||
--hash=sha256:df2a58a472f332225671c35b0a830208b86d004f82baa8530fa3782c85646533 \
|
||||
--hash=sha256:e722f16708d854fe924790e051061f6704a472c3bac347b6fd88033ea8dd0dc5 \
|
||||
--hash=sha256:ecfed7367f965a0328cfbdd70da860f15441f002f613185668c6e6ebf5a0ac11 \
|
||||
--hash=sha256:eeac2acb5a20ed25e0ad6d1df9891a520b78b404266b6d11778f25d5d691a6c9 \
|
||||
--hash=sha256:f59e38625469987d7ef6d495323c55e7db6c212eaf6112267e0d3b565a2e9c9f \
|
||||
--hash=sha256:f89831ef99dd7dd169ab06d63a831adb9e20a87aac6d380266bbda5823349169 \
|
||||
--hash=sha256:fd9192b7b70c573d7f214eb1ae35e00d359f6f5e4b27c7e21e30de1fc6204645
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pyjwt
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
black>=26.3.1
|
||||
darker==2.1.1
|
||||
PyGithub==2.6.1
|
||||
cryptography>=48.0.1
|
||||
cryptography>=50.0.0
|
||||
urllib3>=2.7.0
|
||||
requests>=2.33.0
|
||||
idna>=3.15
|
||||
|
||||
Vendored
+1
-1
Submodule External/fmt updated: 1be298e1bd...c07e2aa4b1.
Vendored
+1
-1
Submodule External/rpmalloc updated: 1d85c246cd...09142d7264.
Vendored
+1
-1
Submodule External/vixl updated: 5f418449c4...20bccdbe04.
@@ -407,6 +407,32 @@ def print_parse_enum_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_affects_codegen_options(options, unnamed_options):
|
||||
output_argloader.write("#ifdef CONFIG_AFFECTSCODEGEN\n")
|
||||
output_argloader.write("#undef CONFIG_AFFECTSCODEGEN\n")
|
||||
|
||||
TotalConfigOptions = 0
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
TotalConfigOptions += 1
|
||||
for op_group, group_vals in unnamed_options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
TotalConfigOptions += 1
|
||||
|
||||
output_argloader.write("constexpr static std::array<bool, {}> Config_AffectsCodeGen = {{{{\n".format(TotalConfigOptions))
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
|
||||
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
|
||||
|
||||
for op_group, group_vals in unnamed_options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
|
||||
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
|
||||
output_argloader.write("}};\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
if (len(sys.argv) < 5):
|
||||
sys.exit()
|
||||
|
||||
@@ -451,4 +477,6 @@ print_parse_jsonloader_options(options);
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
print_affects_codegen_options(options, unnamed_options);
|
||||
|
||||
output_argloader.close()
|
||||
@@ -18,6 +18,7 @@ set(SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/DiskCache.cpp
|
||||
Interface/Core/CodeCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
@@ -69,6 +70,7 @@ set(SRCS
|
||||
Utils/LongJump.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/WorkQueueThread.cpp
|
||||
Utils/Profiler.cpp)
|
||||
|
||||
if (ARCHITECTURE_arm64)
|
||||
@@ -301,6 +303,7 @@ add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
|
||||
if (ENABLE_FEX_ALLOCATOR)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_FEX_ALLOCATOR=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC rpmalloc)
|
||||
target_include_directories(JemallocLibs PRIVATE "${PROJECT_SOURCE_DIR}/include/")
|
||||
endif()
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
|
||||
|
||||
@@ -38,8 +38,16 @@ namespace detail {
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
constexpr static std::array<std::string_view, FEXCore::Config::ConfigOption::CONFIG_MAX> option_names = {
|
||||
#define OPT_BASE(type, group, enum, json, default) #json,
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
} // namespace detail
|
||||
|
||||
std::string_view GetConfigJSONName(FEXCore::Config::ConfigOption option) {
|
||||
return FEXCore::Config::detail::option_names[option];
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
PATH_DATA_DIR_GLOBAL,
|
||||
@@ -48,6 +56,7 @@ enum Paths {
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_CONFIG_TELEMETRY_FOLDER,
|
||||
PATH_CACHE_DIR,
|
||||
PATH_LAST,
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
@@ -64,6 +73,10 @@ void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetCacheDirectory(const std::string_view Path) {
|
||||
Paths[PATH_CACHE_DIR] = Path;
|
||||
}
|
||||
|
||||
const fextl::string& GetTelemetryDirectory() {
|
||||
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
|
||||
if (Path.empty()) {
|
||||
@@ -91,6 +104,10 @@ const fextl::string& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetCacheDirectory() {
|
||||
return Paths[PATH_CACHE_DIR];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
@@ -502,4 +519,43 @@ void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArray
|
||||
}
|
||||
}
|
||||
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
|
||||
#define CONFIG_AFFECTSCODEGEN
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
fextl::string SerializeForCache() {
|
||||
fextl::string Config {};
|
||||
|
||||
auto append_string_triple = [](fextl::string& Config, std::string_view Key, ConfigOption Option, auto Value) {
|
||||
Config.append(Key);
|
||||
Config.append(1, '\0');
|
||||
Config.append(fextl::fmt::format("{}", FEXCore::ToUnderlying(Option)));
|
||||
Config.append(1, '\0');
|
||||
Config.append(fextl::fmt::format("{}", Value));
|
||||
Config.append(1, '\0');
|
||||
};
|
||||
|
||||
const auto SerializeValue = [&Config, append_string_triple]<typename T, ConfigOption Option>(auto ConfigVal, const auto Default) {
|
||||
if (!Config_AffectsCodeGen[FEXCore::ToUnderlying(Option)]) {
|
||||
// Skip everything that the config says doesn't affect codegen.
|
||||
return;
|
||||
}
|
||||
append_string_triple(Config, FEXCore::Config::GetConfigJSONName(Option), Option, ConfigVal());
|
||||
};
|
||||
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
SerializeValue.template operator()<type, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
|
||||
#define OPT_STR(group, enum, json, default) \
|
||||
SerializeValue.template operator()<fextl::string, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Unsupported.
|
||||
#define OPT_STRENUM(group, enum, json, default) // Unsupported.
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
return Config;
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool CheckConfigMatches(std::string_view Config) {
|
||||
// Serialize current config and just check if it matches.
|
||||
return SerializeForCache() == Config;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Config
|
||||
@@ -4,6 +4,7 @@
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
@@ -12,6 +13,7 @@
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
@@ -19,6 +21,7 @@
|
||||
"EnableCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Enable the code caching subsystem"
|
||||
]
|
||||
@@ -26,6 +29,7 @@
|
||||
"EnableLazyCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enable lazy loading of chunks in code caches"
|
||||
]
|
||||
@@ -33,6 +37,7 @@
|
||||
"EnableCodeCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enable expensive validation when loading code caches"
|
||||
]
|
||||
@@ -40,6 +45,8 @@
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
"AffectsCodeGen": "true",
|
||||
"Comment": "Technically affects codegen, but this is serialized elsewhere.",
|
||||
"Enums": {
|
||||
"ENABLESVE": "enablesve",
|
||||
"DISABLESVE": "disablesve",
|
||||
@@ -115,6 +122,7 @@
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Scales the cycle counter on systems that have low frequencies."
|
||||
]
|
||||
@@ -122,6 +130,7 @@
|
||||
"HideHybrid": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Hides hybrid CPU core arrangement."
|
||||
]
|
||||
@@ -129,15 +138,74 @@
|
||||
"CPUFeatureRegisters": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but this is serialized in to HostFeatures.",
|
||||
"Desc": [
|
||||
"Allows overriding cpu feature flags for manual testing"
|
||||
]
|
||||
},
|
||||
"DiskCache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables disk caching for code blocks"
|
||||
]
|
||||
},
|
||||
"DiskCacheFileMapping": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Maps cache files for faster reading"
|
||||
]
|
||||
},
|
||||
"DiskCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Debug mode that does nothing but validate code hits"
|
||||
]
|
||||
},
|
||||
"DiskCacheRelocationFilter": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Don't cache blocks with relocations pointing outside of any known region"
|
||||
]
|
||||
},
|
||||
"DiskCacheAnonCaching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Attempt to cache anonymous code"
|
||||
]
|
||||
},
|
||||
"DiskCachePath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Optional base directory override for disk cache"
|
||||
]
|
||||
},
|
||||
"DiskCacheRODBNames": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Optional list of extra read-only disk cache DBs to consider"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
@@ -152,6 +220,7 @@
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
@@ -159,6 +228,7 @@
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
@@ -166,6 +236,7 @@
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
@@ -180,6 +251,7 @@
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
@@ -187,6 +259,7 @@
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
@@ -196,6 +269,7 @@
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
@@ -203,6 +277,7 @@
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
@@ -211,6 +286,7 @@
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
@@ -219,6 +295,7 @@
|
||||
"DynamicL1CacheIncreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "250",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
|
||||
"Lower numbers means more aggressive scaling upward to the maximum size.",
|
||||
@@ -230,6 +307,7 @@
|
||||
"DynamicL1CacheDecreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "50",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
|
||||
"The higher the number, the more aggressively it reduces the L1 cache size.",
|
||||
@@ -243,6 +321,7 @@
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
@@ -250,6 +329,7 @@
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
@@ -257,6 +337,7 @@
|
||||
"DumpIR": {
|
||||
"Type": "str",
|
||||
"Default": "no",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to dump the IR in to.",
|
||||
"[no, stdout, stderr, server, <Folder>]"
|
||||
@@ -265,6 +346,7 @@
|
||||
"PassManagerDumpIR": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::PassManagerDumpIR::OFF",
|
||||
"AffectsCodeGen": "false",
|
||||
"Enums": {
|
||||
"BEFOREOPT": "beforeopt",
|
||||
"AFTEROPT": "afteropt",
|
||||
@@ -283,6 +365,7 @@
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
@@ -290,6 +373,7 @@
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
@@ -297,6 +381,7 @@
|
||||
"GlobalJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name all JIT state as one symbol",
|
||||
"Useful for querying how much time is spent inside of the JIT",
|
||||
@@ -306,6 +391,7 @@
|
||||
"LibraryJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols grouped by library",
|
||||
"Useful for querying how much time is spent in each guest library",
|
||||
@@ -315,6 +401,7 @@
|
||||
"BlockJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols",
|
||||
"Useful for determining hot blocks of code",
|
||||
@@ -324,6 +411,7 @@
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
@@ -334,6 +422,7 @@
|
||||
"InjectLibSegFault": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Sets the environment variable LD_PRELOAD=libSegFault.so",
|
||||
"This allows the user to very easily enable libSegFault without dealing with environment variables",
|
||||
@@ -345,6 +434,7 @@
|
||||
"Disassemble": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::Disassemble::OFF",
|
||||
"AffectsCodeGen": "false",
|
||||
"Enums": {
|
||||
"DISPATCHER": "dispatcher",
|
||||
"BLOCKS": "blocks",
|
||||
@@ -361,6 +451,7 @@
|
||||
"X86Disassemble": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables x86/x86-64 guest disassembly output for compiled blocks.",
|
||||
"Requires FEX to be built with -DENABLE_ZYDIS=TRUE"
|
||||
@@ -369,6 +460,7 @@
|
||||
"ForceSVEWidth": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Allows overriding the SVE width in the vixl simulator.",
|
||||
"Useful as a debugging feature."
|
||||
@@ -377,6 +469,7 @@
|
||||
"DisableTelemetry": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Disables telemetry at runtime.",
|
||||
"Useful for CI instcountCI mostly"
|
||||
@@ -387,6 +480,7 @@
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
@@ -394,6 +488,7 @@
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stderr, server, <Filename>]"
|
||||
@@ -402,6 +497,7 @@
|
||||
"TelemetryDirectory": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/fex-emu/Telemetry/}"
|
||||
@@ -410,6 +506,7 @@
|
||||
"ProfileStats": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
@@ -418,6 +515,7 @@
|
||||
"EnableGpuvisProfiling": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables profiling when FEX was built with the gpuvis profiler backend."
|
||||
]
|
||||
@@ -427,6 +525,7 @@
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"AffectsCodeGen": "true",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
@@ -439,6 +538,7 @@
|
||||
"TSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Controls TSO IR ops.",
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
@@ -447,6 +547,7 @@
|
||||
"VectorTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
|
||||
]
|
||||
@@ -454,6 +555,7 @@
|
||||
"MemcpySetTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
|
||||
"Only affects REP MOVS and REP STOS instructions"
|
||||
@@ -462,6 +564,7 @@
|
||||
"HalfBarrierTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
@@ -470,6 +573,7 @@
|
||||
"StrictInProcessSplitLocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
@@ -478,6 +582,7 @@
|
||||
"KernelUnalignedAtomicBackpatching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
|
||||
]
|
||||
@@ -485,6 +590,7 @@
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
@@ -493,6 +599,7 @@
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
@@ -500,6 +607,7 @@
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
@@ -508,6 +616,7 @@
|
||||
"HideHypervisorBit": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Hides the hypervisor CPUID bit when set.",
|
||||
"Should only be used for applications that have issues with this set."
|
||||
@@ -516,6 +625,7 @@
|
||||
"StartupSleep": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Sleeps the process at startup for a duration of seconds.",
|
||||
"Useful if an application crashes too quickly to attach a debugger."
|
||||
@@ -524,6 +634,7 @@
|
||||
"StartupSleepProcName": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
@@ -531,6 +642,7 @@
|
||||
"MonoHacks": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
|
||||
]
|
||||
@@ -540,6 +652,7 @@
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
@@ -547,6 +660,7 @@
|
||||
"NeedsSeccomp": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
@@ -554,6 +668,7 @@
|
||||
"ExtendedVolatileMetadata": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
|
||||
"Limited in its use but can be handy.",
|
||||
@@ -578,15 +693,18 @@
|
||||
"Misc": {
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false"
|
||||
},
|
||||
"APP_FILENAME": {
|
||||
"Type": "str",
|
||||
"Default": ""
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false"
|
||||
},
|
||||
"APP_CONFIG_NAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"This is the application config name that has been loaded.",
|
||||
"This differs from APP_FILENAME in two ways",
|
||||
@@ -597,16 +715,29 @@
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but this is serialized elsewhere."
|
||||
},
|
||||
"DISABLE_VIXL_INDIRECT_RUNTIME_CALLS": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but only shows up in the test harness.",
|
||||
"Desc": [
|
||||
"This option is used for the InstructionCountCI so it can generate the same codegen between Arm64 hosts and vixl simulator hosts.",
|
||||
"Vixl simulator indirect runtime calls are a special hlt instruction with metadata after it. Effectively making a custom call instruction.",
|
||||
"With visual simulator calls disabled, the code generation would be the same as on a native Arm64 host, but running the code is broken."
|
||||
]
|
||||
},
|
||||
"CONFIG_VERSION": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "true",
|
||||
"Comment": [
|
||||
"Meta option that if config has ever changed definitions dramatically enough that we can rev the version.",
|
||||
"Be mindful that this will invalidate all caches!"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -58,7 +58,7 @@ bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::Interna
|
||||
|
||||
bool FEXCore::Context::ContextImpl::RequiresRelocatableConstants() const {
|
||||
// Support relocation when generating a cache or when generating reference code for validation
|
||||
return CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHEVALIDATION();
|
||||
return CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHEVALIDATION() || DiskCache.IsWritingDiskCache();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -10,6 +10,7 @@
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/DiskCache.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
@@ -121,14 +122,17 @@ public:
|
||||
* Note that FEX relocations are unrelated to ELF/PE relocations.
|
||||
*
|
||||
* @param GuestDelta Guest address offset to apply to RIP-relative data
|
||||
* @param RelocationOffset Offset to subtract from relocation target offsets
|
||||
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
|
||||
*
|
||||
* @return Returns true on success
|
||||
*/
|
||||
[[nodiscard]]
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations,
|
||||
uint32_t RelocationOffset, bool ForStorage);
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
|
||||
|
||||
// Same but on disk cache packed relocations
|
||||
[[nodiscard]]
|
||||
bool ApplyPackedCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
|
||||
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs);
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::SharedCodeBufferManager {
|
||||
@@ -156,32 +160,32 @@ public:
|
||||
void SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - Thread = CreateThread();
|
||||
* - Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
* - Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = InitialStack;
|
||||
* - CTX->ExecuteThread(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread = CreateThread(NewState);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecuteThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - Thread = CreateThread(CopyOfThreadState);
|
||||
* - ExecuteThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread = CreateThread(NewState);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) override;
|
||||
FEXCore::Core::InternalThreadState* CreateThread(const FEXCore::Core::CPUState* NewThreadState) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
@@ -202,6 +206,8 @@ public:
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
virtual void InitDiskCache() override {}
|
||||
|
||||
CodeCache& GetCodeCache() override {
|
||||
return CodeCache;
|
||||
}
|
||||
@@ -243,6 +249,9 @@ public:
|
||||
}
|
||||
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
std::atomic<uint64_t>& GetMonoBackPatcherBlock() {
|
||||
return MonoBackpatcherBlock;
|
||||
}
|
||||
|
||||
// Manual debugging tooling which is useful for developers.
|
||||
struct TrackingEmpty {
|
||||
@@ -375,6 +384,7 @@ public:
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
DiskCache::DiskCache DiskCache;
|
||||
CodeCache CodeCache;
|
||||
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
|
||||
|
||||
|
||||
@@ -1123,6 +1123,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
LOGMAN_THROW_A_FMT((CurrentOffset & 3) == 0, "Can't Align16B code that isn't 4-byte aligned!");
|
||||
for (uint64_t i = (-CurrentOffset & 0xF); i != 0; i -= 4) {
|
||||
nop();
|
||||
}
|
||||
|
||||
@@ -307,7 +307,7 @@ namespace CPU {
|
||||
|
||||
CPUBackend::~CPUBackend() = default;
|
||||
|
||||
auto CPUBackend::GetEmptySharedCodeBuffer() -> CodeBuffer* {
|
||||
auto CPUBackend::AcquireNewSharedCodeBuffer() -> CodeBuffer* {
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
@@ -339,11 +339,8 @@ namespace CPU {
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
const auto CheckCodeBuffer = [](const CodeBuffer& Buffer, uintptr_t Address) {
|
||||
const auto BufferPtr = reinterpret_cast<uintptr_t>(Buffer.Ptr);
|
||||
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
const uintptr_t LastPageAddr = AlignDown(BufferPtr + Buffer.AllocatedSize - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
const auto BufferPtr = reinterpret_cast<uintptr_t>(Buffer.GetBufferBase());
|
||||
const uintptr_t LastPageAddr = BufferPtr + Buffer.UsableSize();
|
||||
return (Address >= BufferPtr && Address < LastPageAddr);
|
||||
};
|
||||
|
||||
|
||||
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
@@ -56,6 +57,8 @@ namespace CPU {
|
||||
fextl::map<uint64_t, uint8_t*> EntryPoints;
|
||||
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
|
||||
size_t Size;
|
||||
// Offset of BlockBegin from the start of the CodeBuffer it lives in
|
||||
uint64_t HostCodeOffset;
|
||||
};
|
||||
|
||||
// Header that can live at the start of a JIT block.
|
||||
@@ -115,6 +118,10 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
virtual CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) {
|
||||
return {};
|
||||
}
|
||||
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
@@ -138,8 +145,9 @@ namespace CPU {
|
||||
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
// Acquires a new shared code buffer, setting `CurrentCodeBuffer` and returning a pointer to it.
|
||||
[[nodiscard]]
|
||||
CodeBuffer* GetEmptySharedCodeBuffer();
|
||||
CodeBuffer* AcquireNewSharedCodeBuffer();
|
||||
|
||||
// This is the code buffer containing the main code under execution by this thread.
|
||||
// CheckCodeBufferUpdate must be used before compiling new code.
|
||||
|
||||
@@ -493,8 +493,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual 8086 mode enhancements
|
||||
(0 << 2) | // Debugging extensions
|
||||
(0 << 3) | // Page size extension
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extension
|
||||
(1 << 4) | // RDTSC supported
|
||||
(1 << 5) | // MSR supported
|
||||
(1 << 6) | // PAE
|
||||
@@ -1097,7 +1097,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(1 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
@@ -1347,7 +1347,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO != 0}
|
||||
, GetCPUID {GetCPUID_Syscall} {
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
|
||||
@@ -314,7 +314,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
std::ranges::copy(GIT_HASH, header.FEXVersion);
|
||||
header.NumBlocks = LookupCache.BlockList.size();
|
||||
header.NumCodePages = LookupCache.CodePages.size();
|
||||
header.CodeBufferSize = FEXCore::AlignUp(CTX.LatestOffset, Utils::FEX_PAGE_SIZE);
|
||||
header.CodeBufferSize = FEXCore::AlignUp(CodeBuffer->AllocatedSpaceUsed(), Utils::FEX_PAGE_SIZE);
|
||||
header.NumRelocations = Relocations.size();
|
||||
header.SerializedBaseAddress = SerializedBaseAddress;
|
||||
::write(fd, &header, sizeof(header));
|
||||
@@ -337,7 +337,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
|
||||
Guest -= SourceBinary.FileStartVA;
|
||||
::write(fd, &Guest, sizeof(Guest));
|
||||
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
|
||||
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->GetBufferBase());
|
||||
::write(fd, &HostCode, sizeof(HostCode));
|
||||
uint64_t NumCodePages = Host->CodePages.size();
|
||||
::write(fd, &NumCodePages, sizeof(NumCodePages));
|
||||
@@ -361,8 +361,9 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
}
|
||||
|
||||
// Dump the host code (relocated for position-independent serialization)
|
||||
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, 0, true)) {
|
||||
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->GetBufferBase()),
|
||||
reinterpret_cast<std::byte*>(CodeBuffer->GetBufferBase()) + CodeBuffer->AllocatedSpaceUsed());
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
|
||||
return false;
|
||||
}
|
||||
@@ -405,7 +406,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
ERROR_AND_DIE_FMT("Failed to create cache load validation context");
|
||||
}
|
||||
|
||||
ValidationThread.reset(ValidationCTX->CreateThread(0, 0, nullptr));
|
||||
ValidationThread.reset(ValidationCTX->CreateThread(nullptr));
|
||||
|
||||
auto Frame = ValidationThread->CurrentFrame;
|
||||
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &ValidationGDT[0];
|
||||
@@ -426,11 +427,12 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
while (CachedCode.size_bytes() > NewCodeBuffer->UsableSize()) {
|
||||
ValidationCTX->ClearCodeCache(ValidationThread.get());
|
||||
NewCodeBuffer = ValidationCTX->GetLatest();
|
||||
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->AllocatedSize / 1024 / 1024);
|
||||
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->TotalAllocationSize() / 1024 / 1024);
|
||||
}
|
||||
|
||||
std::span<std::byte> CodeBufferRangeRef =
|
||||
std::as_writable_bytes(std::span {NewCodeBuffer->Ptr, NewCodeBuffer->Ptr + NewCodeBuffer->UsableSize()}).subspan(0, CachedCode.size_bytes());
|
||||
std::as_writable_bytes(std::span {NewCodeBuffer->GetBufferBase(), NewCodeBuffer->GetBufferBase() + NewCodeBuffer->UsableSize()})
|
||||
.subspan(0, CachedCode.size_bytes());
|
||||
|
||||
while (!GuestBlocks.empty()) {
|
||||
auto [CompiledBlocks, _, _2, _3, _4] = ValidationCTX->CompileCode(ValidationThread.get(), *GuestBlocks.begin(), 0 /* TODO: Set MaxInst? */);
|
||||
@@ -444,12 +446,12 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
NewRelocations.erase(std::remove_if(NewRelocations.begin(), NewRelocations.end(), [](const CPU::Relocation& Reloc) {
|
||||
return Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL && Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
}));
|
||||
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, 0, false);
|
||||
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, false);
|
||||
|
||||
if (ValidationCTX->LatestOffset <= CodeBufferRangeRef.size()) {
|
||||
if (NewCodeBuffer->AllocatedSpaceUsed() <= CodeBufferRangeRef.size()) {
|
||||
// Reference compilation produced fewer bytes than our cache, so validation is going to fail.
|
||||
// Make sure we don't output any garbage bytes though.
|
||||
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, ValidationCTX->LatestOffset);
|
||||
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, NewCodeBuffer->AllocatedSpaceUsed());
|
||||
}
|
||||
|
||||
auto [Mismatch, _] = std::mismatch(CodeBufferRangeRef.begin(), CodeBufferRangeRef.end(), CachedCode.begin());
|
||||
@@ -500,47 +502,141 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
|
||||
// Reset Context state for next validation
|
||||
ValidationThread->LookupCache->ClearCache(ValidationThread->LookupCache->AcquireWriteLock());
|
||||
ValidationCTX->LatestOffset = 0;
|
||||
NewCodeBuffer->Reset();
|
||||
|
||||
LogMan::Msg::IFmt(" successfully validated cache");
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, uint32_t RelocationOffset, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset >= RelocationOffset, "Invalid relocation offset");
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset - RelocationOffset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset - RelocationOffset);
|
||||
static inline void ApplySymbolLiteralRelocation(ContextImpl& CTX, const CPU::RelocNamedSymbolLiteral::NamedSymbol Symbol,
|
||||
uint64_t GuestEntry, CPU::Arm64Emitter& Emitter, bool ForStorage) {
|
||||
// Generate a literal so we can place it
|
||||
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Symbol);
|
||||
Emitter.dc64(Pointer);
|
||||
}
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
static inline bool
|
||||
ApplyThunkMoveRelocation(ContextImpl& CTX, const IR::SHA256Sum* Symbol, uint32_t RegisterIndex, CPU::Arm64Emitter& Emitter, bool ForStorage) {
|
||||
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(*Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
return true;
|
||||
}
|
||||
|
||||
static inline void ApplyRIPLiteralRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint64_t GuestEntry, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.dc64(GuestEntry + GuestRIP);
|
||||
}
|
||||
|
||||
static inline void
|
||||
ApplyRIPMoveRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint8_t RegisterIndex, uint64_t GuestEntry, CPU::Arm64Emitter& Emitter) {
|
||||
uint64_t Pointer = GuestRIP + GuestEntry;
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableDataRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
uint64_t Value = 0;
|
||||
memcpy(&Value, reinterpret_cast<const void*>(SiteAddress), ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Value, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline int64_t ReadLiveGuestDisplacement(uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
uint64_t Raw = 0;
|
||||
memcpy(&Raw, reinterpret_cast<const void*>(SiteAddress), ValueSize);
|
||||
// manual sign-extension from guest live bytes
|
||||
// 1/2 sizes not permitted in DetectDataMasks currently
|
||||
if (ValueSize == 4) {
|
||||
return (int32_t)Raw;
|
||||
} else {
|
||||
return (int64_t)Raw;
|
||||
}
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableRIPLiteralRelocation(uint64_t SiteAddress, uint8_t ValueSize, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.dc64(SiteAddress + ValueSize + ReadLiveGuestDisplacement(SiteAddress, ValueSize));
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableRIPMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
const uint64_t Target = SiteAddress + ValueSize + ReadLiveGuestDisplacement(SiteAddress, ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyPackedCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
|
||||
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (auto& Reloc : SmallRelocs) {
|
||||
LOGMAN_THROW_A_FMT(Reloc.Offset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Offset);
|
||||
switch ((CPU::RelocationTypes)Reloc.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
// Generate a literal so we can place it
|
||||
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
|
||||
Emitter.dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer,
|
||||
CPU::Arm64Emitter::PadType::DOPAD);
|
||||
ApplySymbolLiteralRelocation(CTX, (CPU::RelocNamedSymbolLiteral::NamedSymbol)Reloc.Named.Symbol, GuestEntry, Emitter, false);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
|
||||
ApplyRIPLiteralRelocation(CTX, Reloc.RIPLiteral.GuestRIP, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
|
||||
// TODO: Pointers are required to fit within 48-bit VA space.
|
||||
// But forcing 6-byte broke relocations.
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
ApplyRIPMoveRelocation(CTX, Reloc.RIPMove.GuestRIP, Reloc.RIPMove.RegisterIndex, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE: {
|
||||
ApplyPatchableDataRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
|
||||
ApplyPatchableRIPLiteralRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE: {
|
||||
ApplyPatchableRIPMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unknown packed relocation type {}", ToUnderlying((CPU::RelocationTypes)Reloc.Type));
|
||||
}
|
||||
}
|
||||
for (auto& Reloc : ThunkRelocs) {
|
||||
LOGMAN_THROW_A_FMT(Reloc.Offset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Offset);
|
||||
if (!ApplyThunkMoveRelocation(CTX, (const IR::SHA256Sum*)Reloc.SymbolHash, Reloc.RegisterIndex, Emitter, false)) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset);
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
ApplySymbolLiteralRelocation(CTX, Reloc.NamedSymbolLiteral.Symbol, GuestEntry, Emitter, ForStorage);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
if (!ApplyThunkMoveRelocation(CTX, &Reloc.NamedThunkMove.Symbol, Reloc.NamedThunkMove.RegisterIndex, Emitter, ForStorage)) {
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
ApplyRIPLiteralRelocation(CTX, Reloc.GuestRIP.GuestRIP, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
ApplyRIPMoveRelocation(CTX, Reloc.GuestRIP.GuestRIP, Reloc.GuestRIP.RegisterIndex, GuestEntry, Emitter);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -871,7 +967,7 @@ void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte
|
||||
auto StagingSpan = std::span {Staging, Size};
|
||||
for (size_t i = StartPage; i < EndPage; ++i) {
|
||||
auto PageRelocations = SpanPageRelocations(Code, i);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, StagingSpan, PageRelocations, static_cast<uint32_t>(StartOffset), false);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, StagingSpan, PageRelocations, false);
|
||||
Code.LoadedPages[i] = true;
|
||||
}
|
||||
|
||||
@@ -894,7 +990,7 @@ void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte
|
||||
#endif
|
||||
for (size_t i = StartPage; i < EndPage; ++i) {
|
||||
auto PageRelocations = SpanPageRelocations(Code, i);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, 0, false);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, false);
|
||||
Code.LoadedPages[i] = true;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -76,6 +76,9 @@ $end_info$
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#include <arm_acle.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
@@ -103,6 +106,8 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
DiskCache.Init(this);
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
@@ -412,16 +417,12 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) {
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(const FEXCore::Core::CPUState* NewThreadState) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
@@ -481,7 +482,7 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
|
||||
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->AllocatedSize);
|
||||
Symbols.RegisterJITSpace(Buffer->GetBufferBase(), Buffer->TotalAllocationSize());
|
||||
}
|
||||
|
||||
{
|
||||
@@ -541,7 +542,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (!HasCustomIR) {
|
||||
const auto* GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
|
||||
Thread->FrontendDecoder->DecodeLoop(GuestCode);
|
||||
|
||||
const auto* BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
const auto& CodeBlocks = BlockInfo->Blocks;
|
||||
@@ -640,9 +641,28 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
|
||||
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
std::array<uint8_t, 0x10> CodeOriginal;
|
||||
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto crc32 = [](const uint8_t* Ptr, size_t Size) -> uint32_t {
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
uint32_t Result {};
|
||||
#define do_crc(type, suffix) \
|
||||
while (Size >= sizeof(type)) { \
|
||||
Result = __crc32##suffix(Result, *reinterpret_cast<const type*>(Ptr)); \
|
||||
Ptr += sizeof(type); \
|
||||
Size -= sizeof(type); \
|
||||
}
|
||||
do_crc(uint64_t, d);
|
||||
do_crc(uint32_t, w);
|
||||
do_crc(uint16_t, h);
|
||||
do_crc(uint8_t, b);
|
||||
return Result;
|
||||
#else
|
||||
// Unsupported on non-arm.
|
||||
return 0;
|
||||
#endif
|
||||
};
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(
|
||||
Thread->OpDispatcher->Constant(crc32(ExistingCodePtr, DecodedInfo->InstSize)), InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
|
||||
|
||||
@@ -652,7 +672,12 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
|
||||
// Generate a relocatable entry for invalidation purposes.
|
||||
auto EntryReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, 0);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry(EntryReg);
|
||||
|
||||
// Exit the function at this instruction after invalidation.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
@@ -791,6 +816,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
// OpDispatcher IR already released in this case.
|
||||
return {{}, nullptr, 0, 0, false};
|
||||
}
|
||||
@@ -804,6 +831,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
// Raced to compile, release the OpDispatcher IR.
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
.StartAddr = 0,
|
||||
@@ -822,6 +851,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return {
|
||||
.CompiledCode = std::move(CompiledCode),
|
||||
.DebugData = std::move(DebugData),
|
||||
@@ -858,6 +889,47 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, MaxInst);
|
||||
|
||||
std::optional<ExecutableFileSectionInfo> Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
std::optional<DiskCache::CodeHitData> Hit;
|
||||
std::optional<uint64_t> DiskCacheGuestCodeKey;
|
||||
{
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedDiskCacheLookupTime);
|
||||
Hit = DiskCache.Lookup(Thread, Region, GuestRIP, DiskCacheGuestCodeKey);
|
||||
if (Hit && !DiskCache.IsValidating()) {
|
||||
auto LoadedCode = Thread->CPUBackend->LoadCachedCode(Hit->HostCode);
|
||||
if (LoadedCode.BlockBegin) {
|
||||
for (auto& CodePage : Hit->GuestPages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, Hit->EntryPointRIPs, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Hit->EntryPointRIPs.size() == Hit->EntryPointHostOffsets.size(), "Mismatched Disk Cache entrypoint pairs!");
|
||||
|
||||
uintptr_t CachedHostCode = 0;
|
||||
for (size_t i = 0; i < Hit->EntryPointRIPs.size(); i++) {
|
||||
void* HostAddr = LoadedCode.BlockBegin + Hit->EntryPointHostOffsets[i];
|
||||
Thread->LookupCache->AddBlockMapping(Thread, Hit->EntryPointRIPs[i], Hit->GuestPages, HostAddr);
|
||||
if (Hit->EntryPointRIPs[i] == GuestRIP) {
|
||||
CachedHostCode = reinterpret_cast<uintptr_t>(HostAddr);
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(CachedHostCode != 0, "Couldn't find GuestRIP in Disk Cache entrypoints!");
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheHitCount, 1);
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return CachedHostCode;
|
||||
}
|
||||
}
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheMissCount, 1);
|
||||
}
|
||||
|
||||
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
|
||||
|
||||
@@ -870,6 +942,13 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return reinterpret_cast<uintptr_t>(CodePtr);
|
||||
}
|
||||
|
||||
if (DiskCacheGuestCodeKey && Hit && DiskCache.IsValidating()) {
|
||||
DiskCache.Validate(*DiskCacheGuestCodeKey, *Hit, CompiledCode, Region);
|
||||
}
|
||||
|
||||
// if this ever fires, we need to serialize the offset into disk cache
|
||||
LOGMAN_THROW_A_FMT(StartAddr == GuestRIP, "StartAddr offset from GuestRIP");
|
||||
|
||||
// The core managed to compile the code.
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = CompiledCode.BlockBegin;
|
||||
@@ -909,11 +988,6 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
@@ -929,19 +1003,37 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
// Disk Cache
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
if (DiskCacheGuestCodeKey) {
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations;
|
||||
if (DebugData && DebugData->Relocations) {
|
||||
Relocations = *DebugData->Relocations;
|
||||
}
|
||||
std::span<const uint8_t> GuestCode = {reinterpret_cast<const uint8_t*>(StartAddr), Length};
|
||||
const Frontend::Decoder::DecodedBlockInformation* BlockInfo =
|
||||
NeedsAddGuestCodeRanges ? Thread->FrontendDecoder->GetDecodedBlockInfo() : nullptr;
|
||||
DiskCache.Store(Thread, Region, GuestRIP, *DiskCacheGuestCodeKey, GuestCode, CompiledCode, Relocations, BlockInfo);
|
||||
}
|
||||
|
||||
if (CodeMapWriter && Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
|
||||
}
|
||||
|
||||
if (CodeMapWriter) {
|
||||
auto Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
if (Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
// Clear any relocations that might have been generated
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
@@ -954,6 +1046,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, 1);
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, _] = CompileCode(Thread, GuestRIP, 1);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -69,7 +69,6 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
: Thread {Thread}
|
||||
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
|
||||
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
@@ -89,6 +88,11 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Check for wraparound
|
||||
if (Address + Size < Address) {
|
||||
return false;
|
||||
}
|
||||
|
||||
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
|
||||
ExecutableRangeBase = RangeInfo.Base;
|
||||
@@ -110,9 +114,8 @@ bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
std::optional<uint8_t> Byte = PeekByte(0);
|
||||
if (!Byte) {
|
||||
if (!Byte || InstructionSize == MAX_INST_SIZE) {
|
||||
HitNonExecutableRange = true;
|
||||
// Pretend we read 0, the main decode loop will see HitNonExecutableRange and rollback the instruction.
|
||||
return 0;
|
||||
@@ -137,6 +140,8 @@ std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
|
||||
|
||||
uint64_t Res = 0;
|
||||
uint64_t Address = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize);
|
||||
LastFieldReadOffset = (uint8_t)InstructionSize;
|
||||
LastFieldReadSize = Size;
|
||||
if (CheckRangeExecutable(Address, Size)) {
|
||||
std::memcpy(&Res, &InstStream.AdjustedInstStream[InstructionSize], Size);
|
||||
} else {
|
||||
@@ -634,10 +639,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
|
||||
if (Bytes <= 8 && Bytes > 0) {
|
||||
auto [Literal, IsRelocation] = ReadData(Bytes);
|
||||
if (IsRelocation) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::LiteralRelocation;
|
||||
@@ -662,6 +664,11 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
} else {
|
||||
// All real x86 instructions have byte sizes that are 8-bytes or less.
|
||||
// Thunk instruction has an additional 32-byte SHA256 payload that needs to be accounted for.
|
||||
InstructionSize += Bytes;
|
||||
Bytes = 0;
|
||||
}
|
||||
|
||||
@@ -1329,6 +1336,13 @@ void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
.BlockStatus = BlockIt->BlockStatus,
|
||||
};
|
||||
|
||||
if (BlockIt->DataMasks.size()) {
|
||||
auto MaskIt = std::lower_bound(BlockIt->DataMasks.begin(), BlockIt->DataMasks.end(), SplitAddr,
|
||||
[](const DataMask& Mask, uint64_t Addr) { return Mask.FieldAddress < Addr; });
|
||||
SplitBlock.DataMasks.assign(MaskIt, BlockIt->DataMasks.end());
|
||||
BlockIt->DataMasks.erase(MaskIt, BlockIt->DataMasks.end());
|
||||
}
|
||||
|
||||
BlockIt->Size = SplitOffset;
|
||||
BlockIt->NumInstructions = SplitIdx;
|
||||
|
||||
@@ -1353,7 +1367,8 @@ const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 && RIP >= VSyscall_Base && RIP < VSyscall_End) {
|
||||
if (BlockInfo.Is64BitMode && CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Linux && RIP >= VSyscall_Base &&
|
||||
RIP < VSyscall_End) {
|
||||
// VSyscall
|
||||
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
|
||||
// Offset 0: vgettimeofday
|
||||
@@ -1373,106 +1388,140 @@ const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _
|
||||
}
|
||||
|
||||
bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
DecodeInstructionsAtEntry(&Thread, InstStream, PC, MaxInst);
|
||||
SetupDecodeInstructionsAtEntry(&Thread, PC, MaxInst);
|
||||
DecodeLoop(InstStream);
|
||||
bool Uncacheable = HitBadRelocation;
|
||||
DelayedDisownBuffer();
|
||||
return !Uncacheable;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
|
||||
uint64_t TotalInstructions {};
|
||||
|
||||
SectionMinAddress = 0;
|
||||
SectionMaxAddress = ~0ULL;
|
||||
Relocations = nullptr;
|
||||
|
||||
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
|
||||
// If generating cache, attempt to load section bounds and relocations
|
||||
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
|
||||
SectionMinAddress = SectionInfo->FileStartVA;
|
||||
SectionMaxAddress = SectionInfo->EndVA;
|
||||
Relocations = &SectionInfo->FileInfo.Relocations;
|
||||
}
|
||||
void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
|
||||
if (LastFieldReadSize < 4) {
|
||||
return;
|
||||
}
|
||||
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
|
||||
DataMaskType Type;
|
||||
|
||||
// Entry is a jump target
|
||||
BlocksToDecode = {PC};
|
||||
// mov reg,imm
|
||||
if (DecodeInst->OP >= 0xB8 && DecodeInst->OP <= 0xBF) {
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
LiteralToPatch = &Src;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
BlockInfo.CodePages = {CurrentCodePage};
|
||||
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
// we could filter to certain high values that are more likely to be pointers/etc?
|
||||
// const uint64_t Value = Lit->Data.Literal.Value;
|
||||
// if (LiteralToPatch && Value < 0x1000000ULL) {
|
||||
// LiteralToPatch = nullptr;
|
||||
// }
|
||||
Type = DataMaskType::MOV;
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
bool FinalInstruction {false};
|
||||
// jmp/call branches that use a literal rip-relative offset
|
||||
// some of those may be inlined by multiblock and will be cleaned up at decode end
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral()) {
|
||||
LiteralToPatch = &DecodeInst->Src[0];
|
||||
Type = DataMaskType::BRANCH;
|
||||
}
|
||||
|
||||
while (!FinalInstruction && !BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
// todo add a bunch more
|
||||
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
if (LiteralToPatch) {
|
||||
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, Type, LastFieldReadSize});
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
}
|
||||
}
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
void Decoder::PruneInlinedBranchDataMasks() {
|
||||
for (auto& Block : BlockInfo.Blocks) {
|
||||
if (!Block.DataMasks.size()) {
|
||||
continue;
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
const auto& LastInst = Block.DecodedInstructions[Block.NumInstructions - 1];
|
||||
const auto& LastMask = Block.DataMasks.back();
|
||||
|
||||
if (LastMask.Type != DataMaskType::BRANCH) {
|
||||
continue;
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
auto BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
const uint64_t NextInst = LastInst.PC + LastInst.InstSize;
|
||||
if (LastMask.FieldAddress < LastInst.PC || LastMask.FieldAddress + LastMask.ValueSize > NextInst) {
|
||||
continue;
|
||||
}
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
BlockIt->IsEntryPoint = EntryBlock;
|
||||
if (std::ranges::binary_search(BlockInfo.Blocks, NextInst + LastInst.Src[0].Data.LiteralPatchable.Value, std::less {}, &DecodedBlocks::Entry)) {
|
||||
Block.DataMasks.pop_back();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t PCOffset = 0;
|
||||
uint64_t BlockStartOffset = DecodedSize;
|
||||
bool EraseBlock = true; // Unset once the block contains an instruction
|
||||
void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
// counter-intuitively, the masks are also needed for lookup on anon prefix decodes, not just stores
|
||||
bool WantsDataMasks = CTX->DiskCache.IsReadingDiskCache() || CTX->DiskCache.IsWritingDiskCache();
|
||||
// remove this if we ever fixup ValidateCode crc constant after relocations
|
||||
if (CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
WantsDataMasks = false;
|
||||
}
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
while (!FinalInstruction && (Paused || !BlocksToDecode.empty())) {
|
||||
bool Pausing = false;
|
||||
fextl::vector<DecodedBlocks>::iterator BlockIt;
|
||||
if (!Paused || BlockResume == -1) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress == ~0ULL || NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
BlockIt->IsEntryPoint = EntryBlock;
|
||||
|
||||
PCOffset = 0;
|
||||
BlockStartOffset = DecodedSize;
|
||||
EraseBlock = true; // Unset once the block contains an instruction
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
} else if (BlockResume != -1) {
|
||||
BlockIt = BlockInfo.Blocks.begin() + BlockResume;
|
||||
BlockResume = -1;
|
||||
}
|
||||
|
||||
Paused = false;
|
||||
|
||||
while (1) {
|
||||
InstructionSize = 0;
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpAddress = RIPToDecode + PCOffset;
|
||||
auto OpAddress = BlockIt->Entry + PCOffset;
|
||||
auto OpMaxAddress = OpAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
@@ -1494,6 +1543,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
BlockInfo.CodePages.insert(CurrentCodePage);
|
||||
}
|
||||
|
||||
LastFieldReadSize = 0;
|
||||
BlockIt->BlockStatus = DecodeInstruction(OpAddress);
|
||||
if (HitBadRelocation) {
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
@@ -1518,6 +1568,11 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
++BlockIt->NumInstructions;
|
||||
BlockIt->Size += DecodeInst->InstSize;
|
||||
|
||||
// if we weren't provided relocations (guest JIT), try to detect what we can
|
||||
if (WantsDataMasks && BlockIt->BlockStatus == DecodedBlockStatus::SUCCESS && BlockInfo.Is64BitMode && !Relocations) {
|
||||
DetectDataMasks(OpAddress, *BlockIt);
|
||||
}
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (BlockIt->BlockStatus != DecodedBlockStatus::SUCCESS) [[unlikely]] {
|
||||
if (!EntryBlock && BlockIt->BlockStatus != DecodedBlockStatus::BAD_RELOCATION) {
|
||||
@@ -1527,6 +1582,9 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
TotalInstructions -= BlockIt->NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
InstStream -= PCOffset;
|
||||
if (DecodedMaxAddress == OpEndAddress) {
|
||||
DecodedMaxAddress -= PCOffset;
|
||||
}
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
@@ -1540,6 +1598,15 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
break;
|
||||
}
|
||||
|
||||
if (GuestSizePause) {
|
||||
if (GuestSizePause > DecodeInst->InstSize) {
|
||||
GuestSizePause -= DecodeInst->InstSize;
|
||||
} else {
|
||||
GuestSizePause = 0;
|
||||
Pausing = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we need to end the entire multiblock
|
||||
FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (FinalInstruction) {
|
||||
@@ -1552,7 +1619,12 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
|
||||
if (CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions)) {
|
||||
BlockIt->ForceFullSMCDetection = true;
|
||||
// todo abandon patching this for now, as the crc will fail and it will lock up redoing it over and over
|
||||
// we should fix the crc at relocation if this is important
|
||||
BlockIt->DataMasks.clear();
|
||||
}
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
@@ -1561,6 +1633,17 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
PCOffset += DecodeInst->InstSize;
|
||||
InstStream += DecodeInst->InstSize;
|
||||
|
||||
if (Pausing) {
|
||||
Pausing = false;
|
||||
Paused = true;
|
||||
BlockResume = BlockIt - BlockInfo.Blocks.begin();
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Paused) {
|
||||
break;
|
||||
}
|
||||
|
||||
// NOTE: BlockIt is only valid here in the EraseBlock case
|
||||
@@ -1572,6 +1655,16 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
CurrentBlockTargets.clear();
|
||||
EntryBlock = false;
|
||||
|
||||
if (Pausing && !BlocksToDecode.empty() && !FinalInstruction) {
|
||||
Paused = true;
|
||||
BlockResume = -1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Paused) {
|
||||
return;
|
||||
}
|
||||
|
||||
BlockInfo.TotalInstructionCount = TotalInstructions;
|
||||
@@ -1579,6 +1672,65 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
for (auto& Block : BlockInfo.Blocks) {
|
||||
Block.IsEntryPoint = BlockInfo.EntryPoints.contains(Block.Entry);
|
||||
}
|
||||
|
||||
// now that multiblock has settled down, remove any branch masks we put down that didn't end the block
|
||||
if (WantsDataMasks) {
|
||||
PruneInlinedBranchDataMasks();
|
||||
}
|
||||
}
|
||||
|
||||
void Decoder::SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
Paused = false;
|
||||
BlockResume = -1;
|
||||
DecodedSize = 0;
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
this->MaxInst = MaxInst;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
|
||||
TotalInstructions = 0;
|
||||
|
||||
SectionMinAddress = 0;
|
||||
SectionMaxAddress = ~0ULL;
|
||||
Relocations = nullptr;
|
||||
|
||||
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
|
||||
// If generating cache, attempt to load section bounds and relocations
|
||||
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
|
||||
SectionMinAddress = SectionInfo->FileStartVA;
|
||||
SectionMaxAddress = SectionInfo->EndVA;
|
||||
Relocations = &SectionInfo->FileInfo.Relocations;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
|
||||
// Entry is a jump target
|
||||
BlocksToDecode = {PC};
|
||||
|
||||
CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
BlockInfo.CodePages = {CurrentCodePage};
|
||||
|
||||
EntryBlock = true;
|
||||
FinalInstruction = false;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -19,9 +19,6 @@
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::HLE {
|
||||
enum class SyscallOSABI;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
class Decoder final {
|
||||
@@ -35,6 +32,14 @@ public:
|
||||
UNIMPLEMENTED_INST,
|
||||
};
|
||||
|
||||
enum class DataMaskType : uint8_t { MOV, BRANCH };
|
||||
|
||||
struct DataMask final {
|
||||
uint64_t FieldAddress;
|
||||
DataMaskType Type;
|
||||
uint8_t ValueSize;
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
struct DecodedBlocks final {
|
||||
uint64_t Entry {};
|
||||
@@ -44,6 +49,7 @@ public:
|
||||
DecodedBlockStatus BlockStatus;
|
||||
bool IsEntryPoint {};
|
||||
bool ForceFullSMCDetection {};
|
||||
fextl::vector<DataMask> DataMasks;
|
||||
};
|
||||
|
||||
struct DecodedBlockInformation final {
|
||||
@@ -57,7 +63,8 @@ public:
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
bool CheckIfCacheable(FEXCore::Core::InternalThreadState&, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
void SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeLoop(const uint8_t* InstStream, uint64_t GuestPause = 0);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
@@ -74,6 +81,10 @@ public:
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
PoolObject.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
void ResetExecutableRangeCache() {
|
||||
ExecutableRangeBase = ExecutableRangeEnd = 0;
|
||||
}
|
||||
@@ -89,7 +100,6 @@ private:
|
||||
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI {};
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
@@ -102,6 +112,9 @@ private:
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
|
||||
void DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block);
|
||||
void PruneInlinedBranchDataMasks();
|
||||
|
||||
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
|
||||
|
||||
uint8_t ReadByte();
|
||||
@@ -121,6 +134,19 @@ private:
|
||||
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
|
||||
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
uint64_t TotalInstructions {};
|
||||
uint64_t CurrentCodePage {};
|
||||
bool EntryBlock {};
|
||||
bool FinalInstruction {};
|
||||
uint64_t MaxInst {};
|
||||
bool Paused {};
|
||||
int64_t BlockResume = -1;
|
||||
uint64_t PCOffset {};
|
||||
uint64_t BlockStartOffset {};
|
||||
bool EraseBlock {};
|
||||
|
||||
uint8_t LastFieldReadOffset;
|
||||
uint8_t LastFieldReadSize;
|
||||
|
||||
uint64_t ExecutableRangeBase {};
|
||||
uint64_t ExecutableRangeEnd {};
|
||||
@@ -155,6 +181,7 @@ private:
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize {};
|
||||
// Contains the full decoded instruction, unless it is a `Thunk` instruction.
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
uint8_t LastEscapePrefix {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
@@ -10,11 +10,6 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_UNKNOWN, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint64_t* ABIHandlers) {
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = {ABIHandlers[FABI_F80_I16_F32_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4)};
|
||||
@@ -216,12 +211,6 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
|
||||
@@ -64,6 +64,11 @@ DEF_OP(EntrypointOffset) {
|
||||
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestData) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestData>();
|
||||
InsertGuestPatchableDataMove(GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
// nop
|
||||
}
|
||||
|
||||
@@ -58,7 +58,8 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
|
||||
switch (Lit.MoveABI.Header.Type) {
|
||||
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
|
||||
Lit.MoveABI.Header.Offset = GetCursorOffset();
|
||||
break;
|
||||
}
|
||||
@@ -102,6 +103,48 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
auto Arm64JITCore::InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize) -> NamedSymbolLiteralPair {
|
||||
return {
|
||||
.Lit = GuestRIP,
|
||||
.MoveABI =
|
||||
{
|
||||
.GuestPatchableData = {.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL,
|
||||
},
|
||||
.RegisterIndex = 0, // unused
|
||||
.ValueSize = ValueSize,
|
||||
// NOTE: Cache serialization will subtract the unit entry address later
|
||||
.SiteAddress = SiteAddress},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
|
||||
// this might get patched on disk cache load
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
|
||||
// this might get patched on disk cache load
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
|
||||
// Rebase relocations to library base address
|
||||
for (auto& Relocation : Relocations) {
|
||||
|
||||
@@ -83,7 +83,11 @@ DEF_OP(ExitFunction) {
|
||||
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
if (Op->PatchSiteAddress) {
|
||||
InsertGuestPatchableRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress, Op->PatchSiteSize);
|
||||
} else {
|
||||
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
}
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
@@ -173,6 +177,7 @@ DEF_OP(ExitFunction) {
|
||||
ARMEmitter::ForwardLabel TFUnset;
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
// todo do we need to account for cache patching here?
|
||||
InsertGuestRIPMove(TMP1, NewRIP);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
|
||||
@@ -180,7 +185,7 @@ DEF_OP(ExitFunction) {
|
||||
(void)Bind(&TFUnset);
|
||||
}
|
||||
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call, Op->PatchSiteAddress, Op->PatchSiteSize);
|
||||
(void)Bind(&l_CallReturn);
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
}
|
||||
@@ -277,11 +282,9 @@ DEF_OP(CondJump) {
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X0: SyscallHandler
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
@@ -300,31 +303,18 @@ DEF_OP(Syscall) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
// SP supporting move
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
// Syscall result is in any static register that the frontend desired.
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
@@ -337,14 +327,6 @@ DEF_OP(Syscall) {
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -379,53 +361,61 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
auto OldCode = Op->CodeOriginal.data();
|
||||
auto Base = GetReg(Op->Header.Args[0]).X();
|
||||
auto Base = GetReg(Op->Address).X();
|
||||
int len = Op->CodeLength;
|
||||
int Offset = 0;
|
||||
ARMEmitter::ForwardLabel Fail;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto CRC32Reg = GetReg(Op->crc);
|
||||
|
||||
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
|
||||
while (len >= Size) {
|
||||
LoadData();
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
len -= Size;
|
||||
Offset += Size;
|
||||
}
|
||||
};
|
||||
// Changes to TMP1
|
||||
auto WorkingReg = ARMEmitter::XReg::zr;
|
||||
auto BaseReg = TMP2;
|
||||
auto TmpDataReg = TMP3;
|
||||
mov(ARMEmitter::Size::i64Bit, BaseReg, Base);
|
||||
|
||||
EmitCheck(8, [&]() {
|
||||
ldr(TMP1, Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 8) {
|
||||
ldr<ARMEmitter::IndexType::POST>(TmpDataReg, BaseReg, 8);
|
||||
crc32x(TMP1, WorkingReg, TmpDataReg);
|
||||
len -= 8;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
EmitCheck(4, [&]() {
|
||||
ldr(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 4) {
|
||||
ldr<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 4);
|
||||
crc32w(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 4;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
EmitCheck(2, [&]() {
|
||||
ldrh(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 2) {
|
||||
ldrh<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 2);
|
||||
crc32h(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 2;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
EmitCheck(1, [&]() {
|
||||
ldrb(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
|
||||
});
|
||||
while (len >= 1) {
|
||||
ldrb<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 1);
|
||||
crc32b(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 1;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
sub(ARMEmitter::Size::i32Bit, Dst, TMP1, CRC32Reg);
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
cbz_OrRestart(ARMEmitter::Size::i32Bit, Dst, &End);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
auto Op = IROp->C<IR::IROp_ThreadRemoveCodeEntry>();
|
||||
|
||||
// Move the entry to ABI before saving state.
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Entry));
|
||||
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
@@ -434,9 +424,6 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X1: RIP
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
|
||||
// TODO: Relocations don't seem to be wired up to this...?
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry, CPU::Arm64Emitter::PadType::AUTOPAD);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
|
||||
@@ -557,26 +557,30 @@ DEF_OP(Vector_F64ToI32) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// This has a known precision issue that isn't easily resolvable without throwing away performance.
|
||||
// Doing the conversion in multi-stage steps has an issue that you can lose precision in the f32->i32 step if your source was f64.
|
||||
// To get around this with ASIMD FEX needs to use fcvtzs (Scalar, Integer, to GPR) for each F64 to be directly converted to i32.
|
||||
// This is a very costly transform that the SVE path doesn't need to do since it supports f64->i32 directly.
|
||||
// If this precision issue is necessary then we can add an option for it in the future.
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
///< skip TowardsZero as fcvtzs below already truncates toward zero on its own
|
||||
auto CVTReg = Dst.Q();
|
||||
switch (Round) {
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Q(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
// Now narrow from f64 to f32.
|
||||
fcvtn(ARMEmitter::SubRegSize::i32Bit, Dst.Q(), Dst.Q());
|
||||
///< Convert f64 directly to i64
|
||||
fcvtzs(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), CVTReg);
|
||||
|
||||
///< Convert the two F32 integrals to real integers.
|
||||
fcvtzs(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
///< Saturating narrow i64 -> i32
|
||||
///
|
||||
///< The caller(Vector_CVT_Float_To_Int32Impl) only fixes up positive overflow:
|
||||
///< it tests MaxF > Src (MaxF = 2^31) and swaps in CVTMAX_I32 (0x80000000) where
|
||||
///< the test fails.
|
||||
///
|
||||
///< Sources below INT32_MIN are handled by sqxtn:
|
||||
///< ARM saturates to INT32_MIN, which is 0x80000000 the same value as
|
||||
///< x86's integer-indefinite value.
|
||||
sqxtn(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -616,11 +616,11 @@ void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(*ctx, Thread)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128 != 0}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256 != 0}
|
||||
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES != 0}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP != 0}
|
||||
, CTX {ctx}
|
||||
, TempCodeBufferAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
|
||||
@@ -675,10 +675,8 @@ void Arm64JITCore::ClearCache() {
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
|
||||
|
||||
auto CodeBuffer = GetEmptySharedCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->AllocatedSize);
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
|
||||
auto CodeBuffer = AcquireNewSharedCodeBuffer();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -815,6 +813,33 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
|
||||
|
||||
CodeBuffer::CodeBufferAllocation Arm64JITCore::AllocateCodeBufferInSharedCache(size_t Size) {
|
||||
CodeBuffer::CodeBufferAllocation AllocatedInfo {};
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
// Bring CodeBuffer up to date
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// Attempt to allocate a buffer from the SharedCodeBuffers.
|
||||
while (AllocatedInfo.BufferAllocationOffset == nullptr) {
|
||||
AllocatedInfo = CurrentCodeBuffer->AtomicAllocateBuffer(Size);
|
||||
|
||||
if (AllocatedInfo.BufferAllocationOffset == nullptr) {
|
||||
// If it didn't fit then clear the buffer and try again.
|
||||
// This has the possibility of migrating the SharedCodeBuffer. See above in `Arm64JITCore::ClearCache()`
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return AllocatedInfo;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
@@ -867,6 +892,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
LOGMAN_THROW_A_FMT(GetCursorOffset() == 0, "Needs to be zero");
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
@@ -993,9 +1019,15 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
dc64(0); // HostCode
|
||||
if (PendingJumpThunk.PatchSiteAddress) {
|
||||
// GuestRIP with an extra step
|
||||
PlaceNamedSymbolLiteral(
|
||||
InsertGuestPatchableRIPLiteral(PendingJumpThunk.GuestRIP, PendingJumpThunk.PatchSiteAddress, PendingJumpThunk.PatchSiteSize));
|
||||
} else {
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
}
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
BindOrRestart(&l_ExitLink);
|
||||
@@ -1055,66 +1087,44 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
|
||||
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
|
||||
Align();
|
||||
// Make sure code is 16B aligned on the tail.
|
||||
// Can't use Align16B here as vl64pair can cause non-4byte alignment.
|
||||
Align(16);
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
// Beginning of emission is guaranteed to be offset zero. So the code data size is just the current cursor offset.
|
||||
CodeData.Size = GetCursorOffset();
|
||||
|
||||
// Finalize and write block tail data
|
||||
JITBlockTail.Size = CodeData.Size;
|
||||
{
|
||||
auto PrevCur = GetCursorOffset();
|
||||
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
|
||||
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
|
||||
SetCursorOffset(PrevCur);
|
||||
|
||||
// Emitter buffer is no longer used, guard against misuse by setting to nullptr.
|
||||
SetBuffer(nullptr, 0);
|
||||
}
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
{
|
||||
auto CodeBufferLock = std::unique_lock {SharedCodeBuffers.CodeBufferWriteMutex};
|
||||
LOGMAN_THROW_A_FMT(CodeData.Size % 16 == 0, "Needs to be 16B aligned!");
|
||||
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->AllocatedSize);
|
||||
SetCursorOffset(SharedCodeBuffers.LatestOffset);
|
||||
Align16B();
|
||||
if ((GetCursorOffset() + TempSize) > CurrentCodeBuffer->UsableSize()) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
SharedCodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
auto AllocatedInfo = AllocateCodeBufferInSharedCache(CodeData.Size);
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
LOGMAN_THROW_A_FMT((reinterpret_cast<uintptr_t>(AllocatedInfo.BufferAllocationOffset) % 16) == 0, "Allocated buffer wasn't 16B "
|
||||
"aligned?");
|
||||
|
||||
// Adjust host addresses
|
||||
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
const auto Delta = AllocatedInfo.BufferAllocationOffset - CodeData.BlockBegin;
|
||||
CodeData.BlockBegin += Delta;
|
||||
for (auto& EntryPoint : CodeData.EntryPoints) {
|
||||
EntryPoint.second += Delta;
|
||||
}
|
||||
CodeBegin += Delta;
|
||||
|
||||
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
|
||||
Relocations[Idx].Header.Offset += SharedCodeBuffers.LatestOffset;
|
||||
}
|
||||
CodeData.HostCodeOffset = CodeData.BlockBegin - CurrentCodeBuffer->GetBufferBase();
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(SharedCodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
SharedCodeBuffers.LatestOffset = GetCursorOffset();
|
||||
memcpy(AllocatedInfo.BufferAllocationOffset, TempCodeBuffer, CodeData.Size);
|
||||
}
|
||||
|
||||
TempCodeBufferAllocator.DelayedDisownBuffer();
|
||||
@@ -1153,6 +1163,22 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
return std::move(CodeData);
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span<const uint8_t> HostBytes) {
|
||||
// we stored it aligned, better still be?
|
||||
LOGMAN_THROW_A_FMT(HostBytes.size() % 16 == 0, "Needs to be 16B aligned!");
|
||||
auto AllocatedInfo = AllocateCodeBufferInSharedCache(HostBytes.size());
|
||||
|
||||
uint8_t* Dest = AllocatedInfo.BufferAllocationOffset;
|
||||
memcpy(Dest, HostBytes.data(), HostBytes.size());
|
||||
ClearICache(Dest, HostBytes.size());
|
||||
|
||||
CPUBackend::CompiledCode Result;
|
||||
Result.BlockBegin = Dest;
|
||||
Result.Size = HostBytes.size();
|
||||
Result.HostCodeOffset = Dest - CurrentCodeBuffer->GetBufferBase();
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0) {
|
||||
return;
|
||||
|
||||
@@ -54,6 +54,9 @@ public:
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
void ClearRelocations() override {
|
||||
@@ -102,6 +105,8 @@ private:
|
||||
uint64_t CallerAddress;
|
||||
uint64_t GuestRIP;
|
||||
ARMEmitter::ForwardLabel Label;
|
||||
uint64_t PatchSiteAddress = 0;
|
||||
uint8_t PatchSiteSize = 0;
|
||||
};
|
||||
fextl::vector<PendingJumpThunk> PendingJumpThunks;
|
||||
|
||||
@@ -342,8 +347,8 @@ private:
|
||||
uint32_t End;
|
||||
};
|
||||
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call, uint64_t PatchSiteAddress = 0, uint8_t PatchSiteSize = 0) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}, PatchSiteAddress, PatchSiteSize});
|
||||
auto& Thunk = PendingJumpThunks.back();
|
||||
BindOrRestart(&Thunk.Label);
|
||||
if (Call) {
|
||||
@@ -559,6 +564,9 @@ private:
|
||||
*/
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
|
||||
void InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
void InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
@@ -578,6 +586,11 @@ private:
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Like InsertGuestRIPLiteral, but with patch information to recompute value from live guest bytes at cache load time
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
@@ -624,6 +637,8 @@ private:
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
[[nodiscard]] CodeBuffer::CodeBufferAllocation AllocateCodeBufferInSharedCache(size_t Size);
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
///< Unhandled handler
|
||||
|
||||
@@ -2401,13 +2401,13 @@ DEF_OP(CacheLineClear) {
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
if (CTX->HostFeatures.DCacheSize() >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, CurrentWorkingReg);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
|
||||
CurrentWorkingReg = TMP1;
|
||||
}
|
||||
}
|
||||
@@ -2430,13 +2430,13 @@ DEF_OP(CacheLineClean) {
|
||||
|
||||
// Clean dcache only
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
if (CTX->HostFeatures.DCacheSize() >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, CurrentWorkingReg);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
|
||||
CurrentWorkingReg = TMP1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -25,6 +25,18 @@ enum class RelocationTypes : uint32_t {
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocGuestRIP
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
|
||||
// The frontend flagged those regions as patchable by the disk cache
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_DATA_MOVE,
|
||||
|
||||
// Same as GuestRipLiteral but patchable
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_LITERAL,
|
||||
|
||||
// Like PATCHABLE_RIP_LITERAL but puts it in a register
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct FEX_PACKED RelocationHeader final {
|
||||
@@ -73,6 +85,20 @@ struct RelocGuestRIP final {
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
struct RelocGuestPatchableData final {
|
||||
RelocationHeader Header {};
|
||||
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
uint8_t ValueSize;
|
||||
|
||||
char Pad[2];
|
||||
|
||||
uint64_t SiteAddress;
|
||||
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
// Clang 16 Can't default-initialize this union
|
||||
static Relocation Default() {
|
||||
@@ -93,6 +119,8 @@ union Relocation {
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIP GuestRIP;
|
||||
|
||||
RelocGuestPatchableData GuestPatchableData;
|
||||
};
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
|
||||
|
||||
@@ -1297,6 +1297,7 @@ DEF_OP(VFAddP) {
|
||||
const auto Op = IROp->C<IR::IROp_VFAddP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto IsScalar = OpSize == IR::OpSize::i64Bit;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1327,6 +1328,8 @@ DEF_OP(VFAddP) {
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
} else if (IsScalar) {
|
||||
faddp(SubRegSize, Dst.D(), VectorLower.D(), VectorUpper.D());
|
||||
} else {
|
||||
faddp(SubRegSize, Dst.Q(), VectorLower.Q(), VectorUpper.Q());
|
||||
}
|
||||
@@ -3332,6 +3335,55 @@ DEF_OP(VUShrNI2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRSHRN) {
|
||||
const auto Op = IROp->C<IR::IROp_VRSHRN>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
rshrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
} else {
|
||||
rshrn(SubRegSize, Dst.D(), Vector.D(), BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRSHRNPair) {
|
||||
const auto Op = IROp->C<IR::IROp_VRSHRNPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
rshrnb(SubRegSize, VTMP1.Z(), VectorLower.Z(), BitShift);
|
||||
rshrnb(SubRegSize, VTMP2.Z(), VectorUpper.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
} else {
|
||||
if (Dst == VectorUpper) {
|
||||
// RSHRN writes the lower half and would destroy the upper input.
|
||||
mov(VTMP1.Q(), VectorUpper.Q());
|
||||
VectorUpper = VTMP1;
|
||||
}
|
||||
|
||||
rshrn(SubRegSize, Dst.D(), VectorLower.D(), BitShift);
|
||||
rshrn2(SubRegSize, Dst.Q(), VectorUpper.Q(), BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSXTL) {
|
||||
const auto Op = IROp->C<IR::IROp_VSXTL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <span>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
@@ -93,13 +94,15 @@ struct GuestToHostMap {
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second;
|
||||
return BlockList
|
||||
.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, fextl::vector<uint64_t>(CodePages.begin(), CodePages.end())})
|
||||
.first->second;
|
||||
}
|
||||
|
||||
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
|
||||
@@ -220,7 +223,7 @@ public:
|
||||
}
|
||||
|
||||
if (HostPtr && DynamicL1Cache()) {
|
||||
UpdateDynamicL1Stats(Thread);
|
||||
UpdateDynamicL1Stats(Thread, Address, HostPtr);
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
|
||||
@@ -228,7 +231,7 @@ public:
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddress, uint64_t HostCode) {
|
||||
// If host pointer was found in L2 or L3, then add it to the counter.
|
||||
// Keeping track not L1 misses, but specifically L2/L3 hits.
|
||||
++L2L3CacheHits;
|
||||
@@ -242,12 +245,18 @@ public:
|
||||
|
||||
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries < MAX_L1_ENTRIES) {
|
||||
// Entries whose address has the new mask bit set would be unreachable by InvalidateCache
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(L1Pointer), CurrentL1Entries * sizeof(LookupCacheEntry), false);
|
||||
|
||||
CurrentL1Entries <<= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
|
||||
// If L1 was just shrunk, then we just removed our cached entry. Add it back.
|
||||
AddL1Entry(GuestAddress, HostCode);
|
||||
}
|
||||
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries > MIN_L1_ENTRIES) {
|
||||
@@ -275,7 +284,7 @@ public:
|
||||
|
||||
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, auto& Addresses, uint64_t Start, uint64_t Length) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
@@ -285,7 +294,7 @@ public:
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode) {
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
@@ -380,15 +389,19 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
void AddL1Entry(uint64_t GuestAddress, uint64_t HostCode) {
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[GuestAddress & L1PointerMask];
|
||||
L1Entry.GuestCode = GuestAddress;
|
||||
L1Entry.HostCode = HostCode;
|
||||
}
|
||||
|
||||
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
|
||||
for (const auto& CodePage : Entry.CodePages) {
|
||||
CachedCodePages[CodePage >> 12].insert(Address);
|
||||
}
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = Entry.HostCode;
|
||||
AddL1Entry(Address, Entry.HostCode);
|
||||
|
||||
if (!DisableL2Cache() && !L1Only) {
|
||||
// Do ful map
|
||||
|
||||
@@ -36,35 +36,6 @@ using X86Tables::OpToIndex;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
constexpr size_t SyscallArgs = 7;
|
||||
using SyscallArray = std::array<uint64_t, SyscallArgs>;
|
||||
|
||||
size_t NumArguments {};
|
||||
const SyscallArray* GPRIndexes {};
|
||||
static constexpr SyscallArray GPRIndexes_64 = {
|
||||
FEXCore::X86State::REG_RAX, FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_R10, FEXCore::X86State::REG_R8, FEXCore::X86State::REG_R9,
|
||||
};
|
||||
static constexpr SyscallArray GPRIndexes_32 = {
|
||||
FEXCore::X86State::REG_RAX, FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RCX, FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_RBP,
|
||||
};
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64) {
|
||||
NumArguments = GPRIndexes_64.size();
|
||||
GPRIndexes = &GPRIndexes_64;
|
||||
} else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX32) {
|
||||
NumArguments = GPRIndexes_32.size();
|
||||
GPRIndexes = &GPRIndexes_32;
|
||||
} else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// All registers will be spilled before the syscall and filled afterwards so no JIT-side argument handling is necessary.
|
||||
NumArguments = 0;
|
||||
GPRIndexes = nullptr;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unhandled OSABI syscall");
|
||||
}
|
||||
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
@@ -72,13 +43,6 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
auto NewRIP = GetRelocatedPC(Op, -Op->InstSize);
|
||||
_StoreContextGPR(GPRSize, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
|
||||
Ref Arguments[SyscallArgs] {
|
||||
InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode,
|
||||
};
|
||||
for (size_t i = 0; i < NumArguments; ++i) {
|
||||
Arguments[i] = LoadGPRRegister(GPRIndexes->at(i));
|
||||
}
|
||||
|
||||
if (IsSyscallInst) {
|
||||
// If this is the `Syscall` instruction rather than `int 0x80` then we need to do some additional work.
|
||||
// RCX = RIP after this instruction
|
||||
@@ -94,12 +58,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
}
|
||||
|
||||
FlushRegisterCache();
|
||||
auto SyscallOp = _Syscall(Arguments[0], Arguments[1], Arguments[2], Arguments[3], Arguments[4], Arguments[5], Arguments[6]);
|
||||
|
||||
// Generic ABI doesn't store result in RAX.
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
StoreGPRRegister(X86State::REG_RAX, SyscallOp);
|
||||
}
|
||||
_Syscall();
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
@@ -1526,7 +1485,7 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
Res = _Extr(OpSizeFromSrc(Op), Dest, Src, Size - Shift);
|
||||
}
|
||||
|
||||
CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Res, Dest, Shift);
|
||||
CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Res, Dest, Shift, true);
|
||||
CalculateDeferredFlags();
|
||||
StoreResultGPR(Op, Res);
|
||||
} else if (Shift == 0 && Size == 32) {
|
||||
@@ -1774,7 +1733,7 @@ void OpDispatchBuilder::BLSMSKBMIOp(OpcodeArgs) {
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF set according to the Src
|
||||
auto CFInv = To01(OpSize::i64Bit, Src);
|
||||
auto CFInv = To01(Size, Src);
|
||||
|
||||
// The output of BLSMSK is always nonzero, so TST will clear Z (along with C
|
||||
// and O) while setting S.
|
||||
@@ -1792,7 +1751,7 @@ void OpDispatchBuilder::BLSRBMIOp(OpcodeArgs) {
|
||||
|
||||
StoreResultGPR(Op, Result);
|
||||
|
||||
auto CFInv = To01(OpSize::i64Bit, Src);
|
||||
auto CFInv = To01(Size, Src);
|
||||
|
||||
SetNZ_ZeroCV(Size, Result);
|
||||
SetCFInverted(CFInv);
|
||||
@@ -4398,6 +4357,9 @@ AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, con
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.SIB.Base) || IsNonTSOReg(AccessType, Operand.Data.SIB.Index);
|
||||
} else if (Operand.IsLiteralRelocation()) {
|
||||
A.Base = _EntrypointOffset(GPRSize, Operand.Data.LiteralRelocation.EntrypointOffset);
|
||||
} else if (Operand.IsLiteralPatchable()) {
|
||||
A.Base = _PatchableGuestData(OpSize::i64Bit, Operand.Data.LiteralPatchable.Value, Op->PC + Operand.Data.LiteralPatchable.FieldOffset,
|
||||
static_cast<uint64_t>(Operand.Data.LiteralPatchable.Width));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unknown Src Type: {}\n", Operand.Type);
|
||||
}
|
||||
@@ -4617,7 +4579,7 @@ void OpDispatchBuilder::StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref
|
||||
}
|
||||
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9}
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9 != 0}
|
||||
, CTX {ctx}
|
||||
, Thread {Thread} {
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
|
||||
@@ -202,9 +202,10 @@ public:
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, InvalidNode, InvalidNode);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock,
|
||||
uint64_t PatchSiteAddress = 0, uint64_t PatchSiteSize = 0) {
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock);
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock, PatchSiteAddress, PatchSiteSize);
|
||||
}
|
||||
IRPair<IROp_Break> Break(BreakDefinition Reason) {
|
||||
FlushRegisterCache();
|
||||
@@ -360,6 +361,7 @@ public:
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedNoNopOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs, bool IsAVX);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void LSLOp(OpcodeArgs);
|
||||
@@ -799,6 +801,7 @@ public:
|
||||
void VPFCMPOp(OpcodeArgs, uint8_t CompType);
|
||||
void PI2FWOp(OpcodeArgs);
|
||||
void PF2IWOp(OpcodeArgs);
|
||||
void PF2IDOp(OpcodeArgs);
|
||||
|
||||
void PMULHRWOp(OpcodeArgs);
|
||||
|
||||
@@ -1508,18 +1511,24 @@ private:
|
||||
return _GetRelocatedPC(Op, Offset, false);
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
uint64_t PatchOffset = 0;
|
||||
uint64_t PatchSize = 0;
|
||||
if (Op->Src[0].IsLiteralPatchable() && Offset && Offset == (int64_t)Op->Src[0].Literal()) {
|
||||
PatchOffset = Op->PC + Op->Src[0].Data.LiteralPatchable.FieldOffset;
|
||||
PatchSize = Op->Src[0].Data.LiteralPatchable.Width;
|
||||
}
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock, PatchOffset, PatchSize);
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
ExitRelocatedPC(Op, Offset, BranchHint::None, InvalidNode, InvalidNode);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
// Literals are immediates as sources but memory addresses as destinations.
|
||||
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation())) && !Operand.IsGPR();
|
||||
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation() || Operand.IsLiteralPatchable())) && !Operand.IsGPR();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -2131,16 +2140,16 @@ private:
|
||||
}
|
||||
|
||||
// Compares two floats and sets flags for a COMISS instruction
|
||||
void Comiss(IR::OpSize ElementSize, Ref Src1, Ref Src2, bool InvalidateAF = false) {
|
||||
void Comiss(IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
// First, set flags according to Arm FCMP.
|
||||
HandleNZCVWrite();
|
||||
_FCmp(ElementSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
ComissFlags(InvalidateAF);
|
||||
ComissFlags();
|
||||
}
|
||||
|
||||
// Sets flags for a COMISS instruction
|
||||
void ComissFlags(bool InvalidateAF = false) {
|
||||
void ComissFlags() {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
@@ -2155,12 +2164,15 @@ private:
|
||||
Ref V_inv = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(V_inv);
|
||||
|
||||
if (!InvalidateAF) {
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
}
|
||||
// Intel: OF, SF, and AF set to zero
|
||||
// AMD: no mention of OF, SF and AF but actual hardware seems to always zero
|
||||
//
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
// OF and SF are zeroed:
|
||||
// _AXFLAG always produces N=0 (SF), V=0 (OF)
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
|
||||
// Convert NZCV from the Arm representation to an eXternal representation
|
||||
// that's totally not a euphemism for x86, nuh-uh. But maps to exactly we
|
||||
@@ -2372,7 +2384,7 @@ private:
|
||||
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift, bool DoubleWide = false);
|
||||
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
|
||||
@@ -7,7 +7,7 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, false>},
|
||||
{0x1D, 1, &OpDispatchBuilder::PF2IDOp},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
|
||||
@@ -26,8 +26,8 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 2>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
|
||||
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
@@ -35,7 +35,7 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0xB0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 0>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
|
||||
{0xB7, 1, &OpDispatchBuilder::PMULHRWOp},
|
||||
|
||||
{0xBB, 1, &OpDispatchBuilder::PSWAPDOp},
|
||||
|
||||
@@ -432,7 +432,7 @@ void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res) {
|
||||
SetNZP_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift, bool DoubleWide) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -447,8 +447,12 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Re
|
||||
// Extract the last bit shifted in to CF. Shift is already masked, but for
|
||||
// 8/16-bit it might be >= SrcSizeBits, in which case CF is cleared. There's
|
||||
// nothing to do in that case since we already cleared CF above.
|
||||
//
|
||||
// - Double-wide shift has UB when shift is GREATER-THAN operand.
|
||||
// - Single-wide shift has UB when shift is GREATER-THAN-EQUAL operand.
|
||||
const auto SrcSizeBits = IR::OpSizeAsBits(SrcSize);
|
||||
if (Shift < SrcSizeBits) {
|
||||
const bool ShouldSetCF = DoubleWide ? (Shift <= SrcSizeBits) : (Shift < SrcSizeBits);
|
||||
if (ShouldSetCF) {
|
||||
SetCFDirect(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +42,12 @@ void OpDispatchBuilder::MOVVectorUnalignedOp(OpcodeArgs) {
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVVectorUnalignedNoNopOp(OpcodeArgs) {
|
||||
// Moves to same register might have secondary-effects and can't convert to a nop.
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
|
||||
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs, bool IsAVX) {
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
@@ -3583,15 +3589,25 @@ void OpDispatchBuilder::PF2IWOp(OpcodeArgs) {
|
||||
// Float to int32_t
|
||||
Src = _Vector_FToZS(Size, OpSize::i32Bit, Src);
|
||||
|
||||
// We now need to transpose the lower 16-bits of each element together
|
||||
// Only needing to move the upper element down in this case
|
||||
Src = _VUnZip(Size, OpSize::i16Bit, Src, Src);
|
||||
// Truncate the 32-bit integers to 16-bit
|
||||
// Saturate values outside the 16-bit range to smallest and largest 16-bit values
|
||||
Src = _VSQXTN(Size, OpSize::i32Bit, Src);
|
||||
|
||||
// Now we need to sign extend the 16bit value to 32-bit
|
||||
Src = _VSXTL(Size, OpSize::i16Bit, Src);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src, Size);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PF2IDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
Src = _Vector_FToZS(Size, OpSize::i32Bit, Src);
|
||||
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src, Size);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PMULHRWOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
@@ -3776,32 +3792,14 @@ void OpDispatchBuilder::VPMULHWOp(OpcodeArgs, bool Signed) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2) {
|
||||
Ref Res {};
|
||||
if (Size == OpSize::i64Bit) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2);
|
||||
Res = _VSShrI(Size << 1, OpSize::i32Bit, Res, 14);
|
||||
auto OneVector = _VectorImm(Size << 1, OpSize::i32Bit, 1);
|
||||
Res = _VAdd(Size << 1, OpSize::i32Bit, Res, OneVector);
|
||||
return _VUShrNI(Size << 1, OpSize::i32Bit, Res, 1);
|
||||
Ref Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2);
|
||||
return _VRSHRN(Size << 1, OpSize::i32Bit, Res, 15);
|
||||
} else {
|
||||
// 128-bit and 256-bit are less efficient
|
||||
Ref ResultLow;
|
||||
Ref ResultHigh;
|
||||
|
||||
ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2);
|
||||
ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
|
||||
ResultLow = _VSShrI(Size, OpSize::i32Bit, ResultLow, 14);
|
||||
ResultHigh = _VSShrI(Size, OpSize::i32Bit, ResultHigh, 14);
|
||||
auto OneVector = _VectorImm(Size, OpSize::i32Bit, 1);
|
||||
|
||||
ResultLow = _VAdd(Size, OpSize::i32Bit, ResultLow, OneVector);
|
||||
ResultHigh = _VAdd(Size, OpSize::i32Bit, ResultHigh, OneVector);
|
||||
|
||||
// Combine the results
|
||||
Res = _VUShrNI(Size, OpSize::i32Bit, ResultLow, 1);
|
||||
return _VUShrNI2(Size, OpSize::i32Bit, Res, ResultHigh, 1);
|
||||
Ref ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2);
|
||||
Ref ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
return _VRSHRNPair(Size, OpSize::i32Bit, ResultLow, ResultHigh, 15);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -139,9 +139,7 @@ void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
|
||||
void OpDispatchBuilder::FSTToStack(OpcodeArgs) {
|
||||
const uint8_t Offset = Op->OP & 7;
|
||||
if (Offset != 0) {
|
||||
_StoreStackToStack(Offset);
|
||||
}
|
||||
_StoreStackToStack(Offset);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
@@ -172,8 +170,9 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
|
||||
|
||||
// Set Invalid Operation flag if overflow or special value
|
||||
// The x87 exception flags are sticky. Preserve earlier result
|
||||
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(InvalidFlag);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), InvalidFlag));
|
||||
}
|
||||
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
@@ -667,7 +666,6 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
} else {
|
||||
// OF, SF, AF, PF all undefined
|
||||
SetCFDirect(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
@@ -675,10 +673,17 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
|
||||
// Intel: OF, SF, and AF set to zero
|
||||
// AMD: no mention of OF, SF and AF but actual hardware seems to always zero
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_RAW_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
// Set Invalid Operation flag when unordered (NaN comparison)
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
|
||||
// The x87 exception flags are sticky. Preserve earlier result
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), HostFlag_Unordered));
|
||||
|
||||
if (PopTwice) {
|
||||
_PopStackDestroy();
|
||||
@@ -703,7 +708,8 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
|
||||
// Set Invalid Operation flag when unordered (NaN comparison)
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
|
||||
// The x87 exception flags are sticky. Preserve earlier result
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), HostFlag_Unordered));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2) {
|
||||
@@ -719,6 +725,9 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
|
||||
} else {
|
||||
_DecStackTop();
|
||||
}
|
||||
|
||||
// C1 set to 0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
// Operations dealing with loading and storing environment pieces
|
||||
|
||||
@@ -367,7 +367,7 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
} else {
|
||||
HandleNZCVWrite();
|
||||
_F80CmpValue(b);
|
||||
ComissFlags(true /* InvalidateAF */);
|
||||
ComissFlags();
|
||||
}
|
||||
|
||||
if (PopTwice) {
|
||||
|
||||
@@ -34,6 +34,9 @@ CodeBuffer::CodeBuffer(size_t Size)
|
||||
FEXCore::Allocator::VirtualTHPControl(Ptr, Size, FEXCore::Allocator::THPControl::Enable);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
|
||||
CodeBufferEnd = Ptr + UsableSize();
|
||||
CodeBufferOffset = Ptr;
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
@@ -66,7 +69,6 @@ fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::AllocateNew(size_t Size)
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
OnCodeBufferAllocated(Buffer);
|
||||
|
||||
@@ -86,7 +88,7 @@ fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartLargerCodeBuffer() {
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->AllocatedSize;
|
||||
auto NewCodeBufferSize = GetLatest()->TotalAllocationSize();
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
@@ -1,13 +1,14 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
category: Thread shared code buffer management
|
||||
category: code buffer ~ Thread shared code buffer management
|
||||
tags: backend|shared
|
||||
$end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
@@ -20,9 +21,6 @@ struct GuestToHostMap;
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t AllocatedSize; // including guard page; see UsableSize()
|
||||
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
@@ -33,10 +31,73 @@ struct CodeBuffer {
|
||||
|
||||
~CodeBuffer();
|
||||
|
||||
/// Returns the number of bytes available for storing code
|
||||
// Atomically allocate a fixed size buffer out of the current allocated codebuffer.
|
||||
// Lockless because it's just a linear allocator.
|
||||
struct CodeBufferAllocation {
|
||||
const uint8_t* BufferBase;
|
||||
uint8_t* BufferAllocationOffset;
|
||||
};
|
||||
|
||||
CodeBufferAllocation AtomicAllocateBuffer(size_t Size) {
|
||||
Size = FEXCore::AlignUp(Size, 16);
|
||||
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBufferOffset.load()) % 16 == 0, "Buffer needs to always be 16B aligned!");
|
||||
|
||||
auto ExpectedOffset = CodeBufferOffset.load(std::memory_order_relaxed);
|
||||
auto DesiredOffset = ExpectedOffset + Size;
|
||||
|
||||
if (DesiredOffset > CodeBufferEnd) {
|
||||
// Couldn't fit.
|
||||
return {};
|
||||
}
|
||||
|
||||
while (!CodeBufferOffset.compare_exchange_strong(ExpectedOffset, DesiredOffset)) {
|
||||
DesiredOffset = ExpectedOffset + Size;
|
||||
|
||||
if (DesiredOffset > CodeBufferEnd) {
|
||||
// Couldn't fit.
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
// Managed to fit.
|
||||
return {
|
||||
.BufferBase = Ptr,
|
||||
.BufferAllocationOffset = ExpectedOffset,
|
||||
};
|
||||
}
|
||||
|
||||
// Returns the total number of bytes available for storing code
|
||||
size_t UsableSize() const {
|
||||
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
// Returns the full size of the buffer, including the guard page.
|
||||
size_t TotalAllocationSize() const {
|
||||
return AllocatedSize;
|
||||
}
|
||||
|
||||
// Returns the num of bytes currently allocated from the allocator.
|
||||
size_t AllocatedSpaceUsed() const {
|
||||
return CodeBufferOffset - Ptr;
|
||||
}
|
||||
|
||||
// Trivially reset the allocator.
|
||||
void Reset() {
|
||||
CodeBufferOffset = Ptr;
|
||||
}
|
||||
|
||||
// Returns the base of the buffer.
|
||||
uint8_t* GetBufferBase() const {
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
private:
|
||||
uint8_t* Ptr;
|
||||
uint8_t* CodeBufferEnd;
|
||||
size_t AllocatedSize; // including guard page; see UsableSize()
|
||||
|
||||
// Code buffer allocation information.
|
||||
std::atomic<uint8_t*> CodeBufferOffset {};
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -50,6 +111,8 @@ struct CodeBuffer {
|
||||
*/
|
||||
class SharedCodeBufferManager {
|
||||
public:
|
||||
virtual ~SharedCodeBufferManager() = default;
|
||||
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
@@ -62,12 +125,6 @@ public:
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartMaximalCodeBuffer();
|
||||
|
||||
// Write offset into the latest CodeBuffer
|
||||
std::size_t LatestOffset {};
|
||||
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
|
||||
|
||||
private:
|
||||
|
||||
@@ -318,7 +318,7 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x8A, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8B, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x8C, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM, 0}},
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{0x8E, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8F, 1, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_ZERO_REG | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_SF_SRC_RAX, 0}},
|
||||
@@ -362,7 +362,7 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xCA, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xCB, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 1}},
|
||||
{0xCE, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_CE] }}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
|
||||
|
||||
@@ -26,8 +26,8 @@ enum Secondary_LUT {
|
||||
|
||||
constexpr std::array<X86InstInfo[2], ENTRY_MAX> Secondary_ArchSelect_LUT = {{
|
||||
{
|
||||
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
|
||||
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
|
||||
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
|
||||
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
|
||||
},
|
||||
{
|
||||
{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX> } },
|
||||
@@ -303,7 +303,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0x3E, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, sizeof(IR::SHA256Sum)}},
|
||||
#endif
|
||||
};
|
||||
|
||||
|
||||
@@ -128,6 +128,7 @@ struct DecodedOperand {
|
||||
RIPRelativeRelocation,
|
||||
Literal,
|
||||
LiteralRelocation,
|
||||
LiteralPatchable,
|
||||
SIB,
|
||||
SIBRelocation
|
||||
};
|
||||
@@ -159,6 +160,9 @@ struct DecodedOperand {
|
||||
bool IsLiteralRelocation() const {
|
||||
return Type == OpType::LiteralRelocation;
|
||||
}
|
||||
bool IsLiteralPatchable() const {
|
||||
return Type == OpType::LiteralPatchable;
|
||||
}
|
||||
bool IsSIB() const {
|
||||
return Type == OpType::SIB;
|
||||
}
|
||||
@@ -167,7 +171,7 @@ struct DecodedOperand {
|
||||
}
|
||||
|
||||
uint64_t Literal() const {
|
||||
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
|
||||
LOGMAN_THROW_A_FMT(IsLiteral() || IsLiteralPatchable(), "Precondition: must be a literal");
|
||||
return Data.Literal.Value;
|
||||
}
|
||||
|
||||
@@ -196,6 +200,12 @@ struct DecodedOperand {
|
||||
int64_t EntrypointOffset;
|
||||
} LiteralRelocation;
|
||||
|
||||
struct {
|
||||
uint64_t Value;
|
||||
uint8_t Size;
|
||||
uint8_t FieldOffset;
|
||||
uint8_t Width;
|
||||
} LiteralPatchable;
|
||||
struct {
|
||||
int64_t Offset;
|
||||
uint8_t Scale;
|
||||
@@ -409,13 +419,6 @@ namespace InstFlags {
|
||||
constexpr InstFlagType SIZE_256BIT = 0b110;
|
||||
constexpr InstFlagType SIZE_64BITDEF = 0b111; // Default mode is 64bit instead of typical 32bit
|
||||
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY;
|
||||
#else
|
||||
// Syscall ends a block on WIN32 because the instruction can update the CPU's RIP.
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY | FLAGS_BLOCK_END;
|
||||
#endif
|
||||
|
||||
constexpr InstFlagType GetSizeDstFlags(InstFlagType Flags) {
|
||||
return (Flags >> FLAGS_SIZE_DST_OFF) & SIZE_MASK;
|
||||
}
|
||||
|
||||
@@ -195,13 +195,13 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"GPR = ValidateCode Array16:$CodeOriginal, GPR:$Address, u8:$CodeLength": {
|
||||
"GPR = ValidateCode GPR:$crc, GPR:$Address, u8:$CodeLength": {
|
||||
"HasSideEffects": true,
|
||||
"HasDest": true,
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"ThreadRemoveCodeEntry": {
|
||||
"ThreadRemoveCodeEntry GPR:$Entry": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
@@ -312,8 +312,8 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock, i64:$PatchSiteAddress{0}, i64:$PatchSiteSize{0}": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP - optionally patchable from guest bytes for caching"
|
||||
],
|
||||
"Inline": ["Any"],
|
||||
"HasSideEffects": true,
|
||||
@@ -326,11 +326,10 @@
|
||||
"CallbackReturn": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5": {
|
||||
"Syscall": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Dispatches a guest syscall through to the SyscallHandler class"
|
||||
],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"Thunk GPR:$ArgPtr, SHA256Sum:$ThunkNameHash": {
|
||||
@@ -954,6 +953,13 @@
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = PatchableGuestData OpSize:#Size, i64:$Value, i64:$SiteAddress, i64:$SiteSize": {
|
||||
"Desc": ["Loads Value in a patchable way",
|
||||
"On disk cache load the value is patched from live guest bytes at SiteAddress"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = Constant i64:$Constant, ConstPad:$Pad{IR::ConstPad::NoPad}, i32:$MaxBytes{0}": {
|
||||
"Desc": ["Generates a 64bit constant inside of a GPR",
|
||||
"Unsupported to create a constant in FPR"
|
||||
@@ -2037,6 +2043,30 @@
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VRSHRN OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": ["Rounding shift right each element and then narrows to the next lower element size",
|
||||
"Writes result to the bottom half of the destination register, upper half is zeroed"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VRSHRNPair OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
|
||||
"Desc": ["Rounding shift right and narrow a pair of vectors into one result"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up"],
|
||||
"DestSize": "RegisterSize",
|
||||
|
||||
@@ -348,10 +348,6 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, BranchHint Ar
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::ostringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
|
||||
*out << fextl::fmt::format("{:02x}", fmt::join(Arg, ""));
|
||||
}
|
||||
|
||||
void Dump(fextl::ostringstream* out, const IRListView* IR) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
|
||||
@@ -36,6 +36,10 @@ public:
|
||||
DualListData.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
DualListData.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
IRListView ViewIR() {
|
||||
return IRListView(&DualListData);
|
||||
}
|
||||
|
||||
@@ -129,6 +129,10 @@ public:
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
PoolObject.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
private:
|
||||
Utils::PoolBufferWithTimedRetirement<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
@@ -272,7 +272,7 @@ private:
|
||||
Ref LoadStackValueAtOffset_Slow(uint8_t Offset = 0);
|
||||
void StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset = 0, bool SetValid = true);
|
||||
// Update Top value in slow path for a pop
|
||||
void UpdateTopForPop_Slow();
|
||||
void UpdateTopForPop_Slow(bool InvalidateTag = true);
|
||||
void UpdateTopForPush_Slow();
|
||||
// Synchronizes the current simulated stack with the actual values.
|
||||
// Returns a new value for Top, that's synchronized between the simulated stack
|
||||
@@ -573,12 +573,16 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
|
||||
HandleBinopValue(Op64, VFOp64, Op80, DestStackOffset, StackOffset2 != DestStackOffset, StackOffset1, StackNode, Reverse);
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::UpdateTopForPop_Slow() {
|
||||
inline void X87StackOptimization::UpdateTopForPop_Slow(bool InvalidateTag) {
|
||||
const auto PopContainer = [](auto& container) {
|
||||
const auto begin = std::begin(container);
|
||||
std::rotate(begin, std::next(begin), std::end(container));
|
||||
};
|
||||
|
||||
if (InvalidateTag) {
|
||||
SetX87ValidTag(0, false);
|
||||
}
|
||||
|
||||
// Pop the top of the x87 stack
|
||||
GetOffsetTopWithCache_Slow(1);
|
||||
PopContainer(TopOffsetCache);
|
||||
@@ -1037,22 +1041,17 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
const auto* Op = IROp->C<IROp_StoreStackToStack>();
|
||||
auto Offset = Op->StackLocation;
|
||||
|
||||
if (Offset != 0) {
|
||||
auto Value = MigrateToSlowPath_IfInvalid();
|
||||
auto Value = MigrateToSlowPath_IfInvalid();
|
||||
|
||||
// Need to store st0 to stack location - basically a copy.
|
||||
if (SlowPath) {
|
||||
StoreStackValueAtOffset_Slow(LoadStackValueAtOffset_Slow(), Offset);
|
||||
} else {
|
||||
StackData.setTop(*Value, Offset);
|
||||
}
|
||||
// Need to store st0 to stack location - basically a copy.
|
||||
if (SlowPath) {
|
||||
StoreStackValueAtOffset_Slow(LoadStackValueAtOffset_Slow(), Offset);
|
||||
} else {
|
||||
StackData.setTop(*Value, Offset);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_POPSTACKDESTROY: {
|
||||
if (SlowPath) {
|
||||
SetX87ValidTag(0, false);
|
||||
}
|
||||
StackPop();
|
||||
break;
|
||||
}
|
||||
@@ -1073,8 +1072,8 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// Slow path: do actual memory operations
|
||||
Ref ValueTop = LoadStackValue();
|
||||
Ref ValueOffset = LoadStackValue(Offset);
|
||||
StoreStackValue(ValueOffset);
|
||||
StoreStackValue(ValueTop, Offset);
|
||||
StoreStackValue(ValueOffset, 0, true);
|
||||
StoreStackValue(ValueTop, Offset, true);
|
||||
} else {
|
||||
// Fast path: swap complete StackMemberInfo preserving Source metadata
|
||||
StackData.setTop(StackMemberOffset, 0);
|
||||
@@ -1178,7 +1177,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_INCSTACKTOP: {
|
||||
if (SlowPath) {
|
||||
UpdateTopForPop_Slow();
|
||||
UpdateTopForPop_Slow(false);
|
||||
} else {
|
||||
StackData.rotate(false);
|
||||
}
|
||||
|
||||
@@ -317,6 +317,7 @@ VirtualTHPPtr VirtualTHPControl {VirtualTHPNOP};
|
||||
void SetupHooks(size_t PageSize, HookPtrs Ptrs) {
|
||||
VirtualName = Ptrs.VirtualName;
|
||||
VirtualTHPControl = Ptrs.VirtualTHPControl;
|
||||
SetupAllocatorHooks(VirtualName);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
|
||||
#ifdef ENABLE_FEX_ALLOCATOR
|
||||
#include <rpmalloc/rpmalloc.h>
|
||||
#ifndef _WIN32
|
||||
@@ -20,16 +22,32 @@
|
||||
namespace FEXCore::Allocator {
|
||||
using mmap_hook_type = void* (*)(void* addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
using munmap_hook_type = int (*)(void* addr, size_t length);
|
||||
using vma_name_hook_type = void (*)(const char* name, const void* address, size_t size);
|
||||
|
||||
#ifdef ENABLE_FEX_ALLOCATOR
|
||||
typedef void* (*rp_mmap_hook_type)(size_t size, size_t alignment, size_t* offset, size_t* mapped_size);
|
||||
typedef void (*rp_munmap_hook_type)(void* address, size_t offset, size_t mapped_size);
|
||||
typedef void (*vma_name_hook_type)(const char* name, const void* address, size_t size);
|
||||
extern "C" rp_mmap_hook_type rp_mmap_hook;
|
||||
extern "C" rp_munmap_hook_type rp_munmap_hook;
|
||||
extern "C" vma_name_hook_type rp_name_hook;
|
||||
|
||||
#ifndef _WIN32
|
||||
mmap_hook_type fex_mmap_hook = ::mmap;
|
||||
munmap_hook_type fex_munmap_hook = ::munmap;
|
||||
|
||||
static inline void LocalVirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
#ifndef _WIN32
|
||||
static bool Supports {true};
|
||||
if (Supports) {
|
||||
auto Result = prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result == -1) {
|
||||
// Disable any additional attempts.
|
||||
Supports = false;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
// Assume a 64KB page size until told otherwise.
|
||||
@@ -172,6 +190,11 @@ void InitializeAllocator(size_t PageSize) {
|
||||
rpmalloc_initialize_config(&global_interface, &global_config);
|
||||
rp_mmap_hook = FEX_rp_mmap;
|
||||
rp_munmap_hook = FEX_rp_memory_unmap;
|
||||
rp_name_hook = LocalVirtualName;
|
||||
}
|
||||
#else
|
||||
void SetupAllocatorHooks(vma_name_hook_type NameHook) {
|
||||
rp_name_hook = NameHook;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
|
||||
namespace FEXCore::UncheckedLongJump {
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#ifdef __arm64ec__
|
||||
#pragma clang diagnostic ignored "-Winline-asm"
|
||||
#endif
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
__asm volatile(R"(
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Threads {
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread> CreateThread_Default(ThreadFunc Func, void* Arg) {
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread> CreateThread_Default(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags) {
|
||||
ERROR_AND_DIE_FMT("Frontend didn't setup thread creation!");
|
||||
}
|
||||
|
||||
@@ -20,8 +20,8 @@ static FEXCore::Threads::Pointers Ptrs = {
|
||||
.CleanupAfterFork = CleanupAfterFork_Default,
|
||||
};
|
||||
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> FEXCore::Threads::Thread::Create(ThreadFunc Func, void* Arg) {
|
||||
return Ptrs.CreateThread(Func, Arg);
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> FEXCore::Threads::Thread::Create(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags) {
|
||||
return Ptrs.CreateThread(Func, Arg, Flags);
|
||||
}
|
||||
|
||||
void FEXCore::Threads::Thread::CleanupAfterFork() {
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/WorkQueueThread.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
WorkQueueThread::WorkQueueThread(FEXCore::Threads::Flags ThreadFlags) {
|
||||
Thread = FEXCore::Threads::Thread::Create(ThreadEntry, this, ThreadFlags);
|
||||
}
|
||||
|
||||
WorkQueueThread::~WorkQueueThread() {
|
||||
{
|
||||
std::unique_lock lk {Mutex};
|
||||
Stop = true;
|
||||
}
|
||||
CV.notify_one();
|
||||
|
||||
if (Thread && Thread->joinable()) {
|
||||
Thread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void WorkQueueThread::QueueWork(fextl::unique_ptr<WorkItem> Work) {
|
||||
{
|
||||
std::unique_lock lk {Mutex};
|
||||
Queue.push_back(std::move(Work));
|
||||
}
|
||||
CV.notify_one();
|
||||
}
|
||||
|
||||
void WorkQueueThread::ThreadProc() {
|
||||
while (true) {
|
||||
fextl::unique_ptr<WorkItem> Work;
|
||||
{
|
||||
std::unique_lock lk {Mutex};
|
||||
while (!(Stop || !Queue.empty())) {
|
||||
CV.wait(lk);
|
||||
}
|
||||
if (Queue.empty()) {
|
||||
// nothing to do? must be stopping
|
||||
LOGMAN_THROW_A_FMT(Stop, "WorkQueueThread wakes up empty but no Stop?");
|
||||
return;
|
||||
}
|
||||
|
||||
Work = std::move(Queue.front());
|
||||
Queue.pop_front();
|
||||
}
|
||||
|
||||
Work->Run();
|
||||
// Work is destroyed here
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -0,0 +1,584 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <bit>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
/**
|
||||
* A bitset that supports allocating contiguous ranges atomically.
|
||||
* - Lock-free, with the caveat that on contention for > 64-bit it would be faster to acquire a lock.
|
||||
* - If low-contention then atomic-behaviour wins.
|
||||
* - Three modes of allocation:
|
||||
* - 1-bit, single-atomic.
|
||||
* - <= 64-bit, single-atomic, contained within single word (introduces sparsity).
|
||||
* - > 64-bit, multiple-atomic, contiguous, roll-back on contiguous allocation failure.
|
||||
* - Can return failure to allocate even if there is space in certain circumstances.
|
||||
* - If the allocated size crosses multiple words.
|
||||
* - Race to allocation caused contention.
|
||||
* - Remembers last allocation/free for inner-word allocations to improve performance.
|
||||
* - Large greater than atomic-word scans always scan from the start.
|
||||
* - Resetting the bitset with clear() is lower cost when page_size=true.
|
||||
* - MADV_DONTNEED replaces pages with zero-page
|
||||
* - When page_size=false, basic memset is also fairly quick.
|
||||
*/
|
||||
template<bool track_last_allocation = false, bool page_sized = true>
|
||||
class atomic_bitset final {
|
||||
public:
|
||||
void init(void* ptr, size_t bits) {
|
||||
LOGMAN_THROW_A_FMT(bits != 0, "Can't init zero");
|
||||
|
||||
base = reinterpret_cast<uint64_t*>(ptr);
|
||||
bits_to_track = bits;
|
||||
words_to_track = bits / WORD_SIZE_BITS;
|
||||
LOGMAN_THROW_A_FMT(bits % WORD_SIZE_BITS == 0, "Bits to track must match uint64_t");
|
||||
if constexpr (page_sized) {
|
||||
LOGMAN_THROW_A_FMT(bits % (4096 * 8) == 0, "Bits to track must match bit count in page");
|
||||
}
|
||||
|
||||
last_allocation_track.set_last_allocation(0);
|
||||
}
|
||||
|
||||
// Allocate a contiguous buffer of bits.
|
||||
// Returns initial bit offset on success, ~0ULL on failure.
|
||||
size_t allocate(size_t count) {
|
||||
LOGMAN_THROW_A_FMT(count != 0, "Can't allocate zero");
|
||||
LOGMAN_THROW_A_FMT(count <= bits_to_track, "Can't allocate larger than size");
|
||||
|
||||
if (count == 1) [[likely]] {
|
||||
// Common and trivial case.
|
||||
return allocate_one(last_allocation_track.get_last_allocation(), words_to_track);
|
||||
} else if (count <= WORD_SIZE_BITS) [[likely]] {
|
||||
// Allocate up to a single word. Don't allow cross-word allocations
|
||||
// Could cause some sparsity
|
||||
return allocate_inside_word(count, last_allocation_track.get_last_allocation(), words_to_track);
|
||||
}
|
||||
|
||||
// TODO: Always scans from beginning to end.
|
||||
// Support iterative scanning.
|
||||
return allocate_large_amount(count, 0, words_to_track);
|
||||
}
|
||||
|
||||
// Frees a contiguous set of bits.
|
||||
void free(size_t index, size_t count) {
|
||||
LOGMAN_THROW_A_FMT(count != 0, "Can't free zero");
|
||||
LOGMAN_THROW_A_FMT(index < bits_to_track, "Can't free beyond end");
|
||||
|
||||
if (count == 1) [[likely]] {
|
||||
free_one(index);
|
||||
return;
|
||||
} else if ((index % WORD_SIZE_BITS + count) <= WORD_SIZE_BITS) [[likely]] {
|
||||
free_inside_word(index, count);
|
||||
return;
|
||||
}
|
||||
|
||||
free_large_amount(index, count);
|
||||
}
|
||||
|
||||
// Clears the entire bitset.
|
||||
// Not thread safe!
|
||||
void clear() {
|
||||
const size_t bytes = words_to_track * sizeof(uint64_t);
|
||||
if constexpr (page_sized) {
|
||||
// VirtualDontNeed replaces pages with zero page.
|
||||
FEXCore::Allocator::VirtualDontNeed(base, bytes);
|
||||
} else {
|
||||
memset(base, 0, bytes);
|
||||
}
|
||||
|
||||
last_allocation_track.set_last_allocation(0);
|
||||
}
|
||||
|
||||
// Checks if a single bit is set.
|
||||
bool is_set(size_t index) const {
|
||||
const size_t word_index = index / WORD_SIZE_BITS;
|
||||
const size_t word_offset = index % WORD_SIZE_BITS;
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
|
||||
const uint64_t bit_mask = 1ULL << word_offset;
|
||||
return (word_atomic.load() & bit_mask) != 0;
|
||||
}
|
||||
|
||||
size_t size_in_bits() const {
|
||||
return bits_to_track;
|
||||
}
|
||||
|
||||
constexpr static size_t invalid() {
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
// non-atomically returns the number of set bits in the bitset.
|
||||
size_t popcount() const {
|
||||
size_t count {};
|
||||
|
||||
// Just ensure all store are visible.
|
||||
std::atomic_thread_fence(std::memory_order_release);
|
||||
|
||||
for (size_t word_index = 0; word_index < words_to_track; ++word_index) {
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
count += std::popcount(word_atomic.load(std::memory_order_relaxed));
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
private:
|
||||
uint64_t* base {};
|
||||
size_t bits_to_track {};
|
||||
size_t words_to_track {};
|
||||
struct data_to_track_nop {
|
||||
constexpr static size_t get_last_allocation() {
|
||||
return 0;
|
||||
}
|
||||
constexpr static void set_last_allocation(size_t) {}
|
||||
};
|
||||
|
||||
struct data_to_track {
|
||||
std::atomic<size_t> last_allocation_word {};
|
||||
|
||||
constexpr size_t get_last_allocation() const {
|
||||
// It's okay if this isn't up to date, full scan of the region still occurs.
|
||||
return last_allocation_word.load(std::memory_order_relaxed);
|
||||
}
|
||||
constexpr void set_last_allocation(size_t word) {
|
||||
last_allocation_word = word;
|
||||
}
|
||||
};
|
||||
using data_type = typename std::conditional<track_last_allocation, data_to_track, data_to_track_nop>::type;
|
||||
data_type last_allocation_track {};
|
||||
|
||||
constexpr static size_t WORD_SIZE_BITS = sizeof(uint64_t) * 8;
|
||||
|
||||
size_t allocate_one(size_t beginning_word_index, size_t ending_word_index) {
|
||||
// Trivial spin.
|
||||
for (size_t i = beginning_word_index; i < ending_word_index; ++i) {
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[i]);
|
||||
auto expected_word = word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
if (expected_word == ~0ULL) {
|
||||
// Won't pass.
|
||||
continue;
|
||||
}
|
||||
|
||||
// Spin on the word trying to acquire a bit.
|
||||
// Uncontended case should immediately succeed.
|
||||
// Contended case can spin the whole word and lose every acquire.
|
||||
|
||||
do {
|
||||
const auto zero_bit = std::countr_one(expected_word);
|
||||
const auto bit_mask = 1ULL << zero_bit;
|
||||
|
||||
// If mask was already set, then we raced to acquire (returned value will be 1).
|
||||
// If mask not set, then we will have acquired (returned value will be 0).
|
||||
expected_word = word_atomic.fetch_or(bit_mask);
|
||||
if ((expected_word & bit_mask) == 0) {
|
||||
// Acquired the bit, return the offset.
|
||||
last_allocation_track.set_last_allocation(i);
|
||||
return i * WORD_SIZE_BITS + zero_bit;
|
||||
}
|
||||
|
||||
// Bit was already acquired.
|
||||
expected_word |= bit_mask;
|
||||
} while (expected_word != ~0ULL);
|
||||
}
|
||||
|
||||
if constexpr (track_last_allocation) {
|
||||
if (beginning_word_index) {
|
||||
// One more chance to get an allocation.
|
||||
// Scan before the previous allocation to see if any free slots have appeared.
|
||||
return allocate_one(0, beginning_word_index);
|
||||
}
|
||||
}
|
||||
|
||||
// Failure to acquire here.
|
||||
return invalid();
|
||||
}
|
||||
|
||||
size_t allocate_inside_word(size_t count, size_t beginning_word_index, size_t ending_word_index) {
|
||||
for (size_t i = beginning_word_index; i < ending_word_index; ++i) {
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[i]);
|
||||
auto expected_word = word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
// Spin on the word trying to acquire a bit.
|
||||
// Uncontended case should immediately succeed.
|
||||
// Contended case can spin the whole word and lose every acquire.
|
||||
while (expected_word != ~0ULL) {
|
||||
auto zero_bit = std::countr_one(expected_word);
|
||||
bool fits = false;
|
||||
uint64_t bit_mask = count == WORD_SIZE_BITS ? ~0ULL : ((1ULL << count) - 1);
|
||||
for (; (zero_bit + count) <= WORD_SIZE_BITS; ++zero_bit) {
|
||||
// Check if the bits fit.
|
||||
uint64_t tmp_bit_mask = bit_mask << zero_bit;
|
||||
if ((expected_word & tmp_bit_mask) == 0) {
|
||||
fits = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Could never fit, early exit.
|
||||
if (!fits) {
|
||||
break;
|
||||
}
|
||||
|
||||
// Shift bit_mask to the desired location.
|
||||
bit_mask <<= zero_bit;
|
||||
|
||||
uint64_t desired_word {};
|
||||
bool acquired = true;
|
||||
|
||||
do {
|
||||
if (expected_word & bit_mask) {
|
||||
// Couldn't acquire this field, move to the next.
|
||||
acquired = false;
|
||||
break;
|
||||
}
|
||||
|
||||
// We desire setting a single bit.
|
||||
desired_word = expected_word | bit_mask;
|
||||
} while (!word_atomic.compare_exchange_strong(expected_word, desired_word));
|
||||
|
||||
if (acquired) {
|
||||
// Acquired the bit, return the offset.
|
||||
last_allocation_track.set_last_allocation(i);
|
||||
return i * WORD_SIZE_BITS + zero_bit;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if constexpr (track_last_allocation) {
|
||||
if (beginning_word_index) {
|
||||
// One more chance to get an allocation.
|
||||
// Scan before the previous allocation to see if any free slots have appeared.
|
||||
return allocate_inside_word(count, 0, beginning_word_index);
|
||||
}
|
||||
}
|
||||
|
||||
// Failure to acquire here.
|
||||
return invalid();
|
||||
}
|
||||
|
||||
size_t allocate_large_amount(size_t count, size_t beginning_word_index, size_t ending_word_index) {
|
||||
// This version of the code needs to explicitly deal with large allocations that can't fit in a word.
|
||||
// So we are always scanning minimum 2 words.
|
||||
const size_t num_words_to_scan = FEXCore::AlignUpPowerOf2(count, WORD_SIZE_BITS) / WORD_SIZE_BITS;
|
||||
const size_t last_word_to_scan = ending_word_index - num_words_to_scan - 1;
|
||||
|
||||
// Scan forward to find the first word.
|
||||
for (size_t base_index = beginning_word_index; base_index < last_word_to_scan;) {
|
||||
uint64_t leading_zeros {};
|
||||
|
||||
size_t center_word_index = 1;
|
||||
bool has_center {};
|
||||
|
||||
size_t tail_bits {};
|
||||
|
||||
size_t remaining_bits = count;
|
||||
|
||||
auto check_head_fitment = [&]() -> bool {
|
||||
auto base_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
auto base_expected_word = base_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
// Count the leading zeros, if it is above zero then we can start here.
|
||||
leading_zeros = std::countl_zero(base_expected_word);
|
||||
|
||||
if (leading_zeros == 0) {
|
||||
// Nope.
|
||||
++base_index;
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto check_center_fitment = [&]() -> bool {
|
||||
// Subtract the number of zeros.
|
||||
remaining_bits -= leading_zeros;
|
||||
|
||||
has_center = remaining_bits >= WORD_SIZE_BITS;
|
||||
|
||||
while (remaining_bits >= WORD_SIZE_BITS) {
|
||||
// All words in-between head and tail must be zero.
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[base_index + center_word_index]);
|
||||
auto center_expected_word = center_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
if (center_expected_word != 0) {
|
||||
// Couldn't fit, won't ever fit in this range, so jump ahead to the last scanned item.
|
||||
// Might still be able to start on the tail of this word.
|
||||
base_index += center_word_index;
|
||||
return false;
|
||||
}
|
||||
|
||||
++center_word_index;
|
||||
remaining_bits -= WORD_SIZE_BITS;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto check_tail_fitment = [&]() -> bool {
|
||||
tail_bits = remaining_bits;
|
||||
if (tail_bits) {
|
||||
// Now for the tail (if it is necessary).
|
||||
size_t tail_index = center_word_index;
|
||||
|
||||
// Count the trailing zeros, if it fits out remaining bits then we can try and allocate.
|
||||
auto tail_word_atomic = std::atomic_ref<uint64_t>(base[base_index + tail_index]);
|
||||
auto tail_expected_word = tail_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
const auto trailing_zeros = std::countr_zero(tail_expected_word);
|
||||
|
||||
if (trailing_zeros < remaining_bits) {
|
||||
// Couldn't fit, but also won't ever fit in this range. Jump ahead to this tail item.
|
||||
// Might still be able to start on the tail of this word.
|
||||
base_index += tail_index;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
if (!check_head_fitment()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!check_center_fitment()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!check_tail_fitment()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// We can try fitting!
|
||||
if (attempt_allocate_range_from_base(base_index, count, leading_zeros, tail_bits, has_center)) {
|
||||
const size_t head_leading_offset = (WORD_SIZE_BITS - leading_zeros);
|
||||
return base_index * WORD_SIZE_BITS + head_leading_offset;
|
||||
}
|
||||
|
||||
// Failure to fit here means we could never fit in this full range. Jump past all the bits
|
||||
// Rescanning the tail to avoid fragmented sparsity on the tails.
|
||||
base_index += num_words_to_scan - 1;
|
||||
}
|
||||
|
||||
return invalid();
|
||||
}
|
||||
|
||||
bool attempt_allocate_range_from_base(size_t base_index, size_t count, size_t head_bits, size_t tail_bits, bool has_center) {
|
||||
const bool has_tail = tail_bits != 0;
|
||||
|
||||
const uint64_t head_bit_mask = head_bits == WORD_SIZE_BITS ? ~0ULL : (((1ULL << head_bits) - 1) << (WORD_SIZE_BITS - head_bits));
|
||||
const uint64_t tail_bit_mask = (1ULL << tail_bits) - 1;
|
||||
const uint64_t center_words_count = (count - head_bits - tail_bits) / WORD_SIZE_BITS;
|
||||
const uint64_t center_words_base_index = base_index + 1;
|
||||
const uint64_t tail_word_base_index = center_words_base_index + center_words_count;
|
||||
|
||||
// Three distinct sections, all of which need to support rewinding.
|
||||
// - Head: Setting the leading zeros to one
|
||||
// - Always exists. Can be a full word, or partial.
|
||||
// - Center: Setting all in-between words to ~0ULL
|
||||
// - Might not exist if tail is smaller than a word
|
||||
// - Always full words if it does exist.
|
||||
// - Tail: Set all trailing zeros up to the size to 1
|
||||
// - Might not exist if Center perfectly aligned to word edge.
|
||||
// - Always partial words, otherwise it would be considered "Center".
|
||||
|
||||
bool set_head {true};
|
||||
bool set_center {true};
|
||||
bool set_tail {true};
|
||||
size_t num_center_set {};
|
||||
|
||||
// Head first.
|
||||
auto set_head_word = [&]() -> bool {
|
||||
auto head_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
auto head_expected_word = head_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
uint64_t desired_word {};
|
||||
do {
|
||||
if (head_expected_word & head_bit_mask) {
|
||||
// Another thread raced and allocated.
|
||||
return false;
|
||||
}
|
||||
|
||||
// Set the whole mask in one atomic operation.
|
||||
desired_word = head_expected_word | head_bit_mask;
|
||||
} while (!head_word_atomic.compare_exchange_strong(head_expected_word, desired_word));
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto clear_head_word = [&]() {
|
||||
auto head_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
head_word_atomic.fetch_and(~head_bit_mask);
|
||||
};
|
||||
|
||||
auto set_center_words = [&]() -> bool {
|
||||
const size_t end_center_word_index = center_words_base_index + center_words_count;
|
||||
for (size_t center_index = center_words_base_index; center_index < end_center_word_index; ++center_index) {
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[center_index]);
|
||||
auto center_expected_word = center_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
// Center words set a full ~0ULL mask.
|
||||
const uint64_t desired_word {~0ULL};
|
||||
do {
|
||||
if (center_expected_word) {
|
||||
// Another thread raced and allocated.
|
||||
return false;
|
||||
}
|
||||
} while (!center_word_atomic.compare_exchange_strong(center_expected_word, desired_word));
|
||||
|
||||
++num_center_set;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto clear_center_words = [&]() {
|
||||
const size_t end_center_word_index = center_words_base_index + num_center_set;
|
||||
for (size_t center_index = center_words_base_index; center_index < end_center_word_index; ++center_index) {
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[center_index]);
|
||||
center_word_atomic.store(0);
|
||||
}
|
||||
};
|
||||
|
||||
auto set_tail_word = [&]() -> bool {
|
||||
auto tail_word_atomic = std::atomic_ref<uint64_t>(base[tail_word_base_index]);
|
||||
auto tail_expected_word = tail_word_atomic.load(std::memory_order_relaxed);
|
||||
|
||||
uint64_t desired_word {};
|
||||
do {
|
||||
if (tail_expected_word & tail_bit_mask) {
|
||||
// Another thread raced and allocated.
|
||||
return false;
|
||||
}
|
||||
|
||||
// Set the whole mask in one atomic operation.
|
||||
desired_word = tail_expected_word | tail_bit_mask;
|
||||
} while (!tail_word_atomic.compare_exchange_strong(tail_expected_word, desired_word));
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
set_head = set_head_word();
|
||||
|
||||
// Do the center if it exists.
|
||||
if (set_head && has_center) {
|
||||
set_center = set_center_words();
|
||||
}
|
||||
|
||||
// Do the tail if it exists.
|
||||
if (set_head && set_center && has_tail) {
|
||||
set_tail = set_tail_word();
|
||||
}
|
||||
|
||||
if (set_head && set_center && set_tail) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Some stage failed to set, rewind everything.
|
||||
if (set_head) {
|
||||
// Clear the head if it was set
|
||||
clear_head_word();
|
||||
}
|
||||
|
||||
if (has_center && num_center_set) {
|
||||
// `set_center` might not be set, but it still managed to set some of the words.
|
||||
clear_center_words();
|
||||
}
|
||||
|
||||
// Tail doesn't need to rewind as it will never have been set if we got here.
|
||||
return false;
|
||||
}
|
||||
|
||||
#if defined(__ARM_FEATURE_ATOMICS) && __ARM_FEATURE_ATOMICS == 1
|
||||
// Might violate memory-ordering requirements?
|
||||
// TODO: Verify and enable or delete depending.
|
||||
// Provides an 11% (Cortex-X4) to 25% (AmpereOneA) performance improvement.
|
||||
constexpr static bool use_stclr {};
|
||||
static inline void stclr(uint64_t value, uint64_t* addr) {
|
||||
asm volatile("stclrl %[Val], [%[addr]];" ::[Val] "r"(value), [addr] "r"(addr) : "memory");
|
||||
}
|
||||
#endif
|
||||
|
||||
void free_one(size_t index) {
|
||||
const size_t word_index = index / WORD_SIZE_BITS;
|
||||
const size_t word_offset = index % WORD_SIZE_BITS;
|
||||
last_allocation_track.set_last_allocation(word_index);
|
||||
|
||||
const uint64_t bic_bit_mask = 1ULL << word_offset;
|
||||
#if defined(__ARM_FEATURE_ATOMICS) && __ARM_FEATURE_ATOMICS == 1
|
||||
if constexpr (use_stclr) {
|
||||
stclr(bic_bit_mask, &base[word_index]);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
word_atomic.fetch_and(~bic_bit_mask);
|
||||
}
|
||||
|
||||
void free_inside_word(size_t index, size_t count) {
|
||||
const size_t word_index = index / WORD_SIZE_BITS;
|
||||
const size_t word_offset = index % WORD_SIZE_BITS;
|
||||
auto word_atomic = std::atomic_ref<uint64_t>(base[word_index]);
|
||||
last_allocation_track.set_last_allocation(word_index);
|
||||
|
||||
if (count == WORD_SIZE_BITS) {
|
||||
word_atomic.store(0);
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t bic_bit_mask = ((1ULL << count) - 1) << word_offset;
|
||||
#if defined(__ARM_FEATURE_ATOMICS) && __ARM_FEATURE_ATOMICS == 1
|
||||
if constexpr (use_stclr) {
|
||||
stclr(bic_bit_mask, &base[word_index]);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
word_atomic.fetch_and(~bic_bit_mask);
|
||||
}
|
||||
|
||||
void free_large_amount(size_t index, size_t count) {
|
||||
// Incoming count can be less than WORD_SIZE_BITS if it is unaligned and crossing multiple words.
|
||||
// Needs to always handle at minimum a head plus center and/or tail arrangement.
|
||||
const size_t base_index = index / WORD_SIZE_BITS;
|
||||
const size_t last_index = FEXCore::AlignUpPowerOf2(index + count, WORD_SIZE_BITS) / WORD_SIZE_BITS;
|
||||
const size_t num_words_to_scan = last_index - base_index;
|
||||
LOGMAN_THROW_A_FMT(num_words_to_scan > 1, "Needs to be larger than 1 ({}, {})", index, count);
|
||||
|
||||
size_t remaining_bits = count;
|
||||
|
||||
const uint64_t head_offset_start = index % WORD_SIZE_BITS;
|
||||
const uint64_t head_bits = WORD_SIZE_BITS - head_offset_start;
|
||||
|
||||
uint64_t head_mask = head_bits == WORD_SIZE_BITS ? ~0ULL : (((1ULL << head_bits) - 1) << head_offset_start);
|
||||
|
||||
remaining_bits -= head_bits;
|
||||
auto head_word_atomic = std::atomic_ref<uint64_t>(base[base_index]);
|
||||
head_word_atomic.fetch_and(~head_mask);
|
||||
|
||||
const size_t remaining_center_words = remaining_bits / WORD_SIZE_BITS;
|
||||
for (size_t i = 0; i < remaining_center_words; ++i) {
|
||||
// Handle centers if they exist.
|
||||
auto center_word_atomic = std::atomic_ref<uint64_t>(base[base_index + i + 1]);
|
||||
center_word_atomic.store(0);
|
||||
remaining_bits -= WORD_SIZE_BITS;
|
||||
}
|
||||
|
||||
if (remaining_bits) {
|
||||
// Handle tail if they exist, must always be less than WORD_SIZE_BITS.
|
||||
LOGMAN_THROW_A_FMT(remaining_bits < WORD_SIZE_BITS, "Too large");
|
||||
const uint64_t tail_mask = (1ULL << remaining_bits) - 1;
|
||||
|
||||
auto tail_word_atomic = std::atomic_ref<uint64_t>(base[base_index + remaining_center_words + 1]);
|
||||
tail_word_atomic.fetch_and(~tail_mask);
|
||||
}
|
||||
}
|
||||
};
|
||||
} // namespace FEXCore::Utils
|
||||
@@ -0,0 +1,474 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include "Utils/atomic_bitset.h"
|
||||
|
||||
#include <array>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
struct bitmap_allocator_settings {
|
||||
std::array<uint32_t, 3> bucket_sizes;
|
||||
std::array<uint32_t, 3> bucket_allocation_percentages;
|
||||
};
|
||||
|
||||
// Default allocation settings passed through template for easy tinkering.
|
||||
constexpr static bitmap_allocator_settings fex_default_allocator_settings = {
|
||||
.bucket_sizes =
|
||||
{
|
||||
16, // 16B - 1KB
|
||||
512, // 512B - 32KB
|
||||
2048, // 2KB - 128KB
|
||||
},
|
||||
.bucket_allocation_percentages =
|
||||
{
|
||||
// 16B bucket special cased to allocate remaining space left over from other buckets.
|
||||
100,
|
||||
// 40% in the 512B bucket.
|
||||
40,
|
||||
// 10% in the 2KB bucket.
|
||||
10,
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
* A bitmap allocator that is segmented in to three partitioned buckets, with atomic memory allocation.
|
||||
*
|
||||
* This class is strongly coupled with FEXCore::Utils::atomic_bitset to have fast and relatively efficient bitmap allocation in a lock-free
|
||||
* fashion. The bucket granule sizes are roughly calculated to match FEX's needs for typical allocation ranges in a single atomic word. It
|
||||
* supports allocating larger than an atomic word worth of granules with the expectation that those are relatively uncommon. Once a bucket
|
||||
* is full, the allocation has a chance to be allocated in to another bucket at a slightly less efficient space usage This is with the
|
||||
* expectation that we want to keep a buffer around as long as possible, without allocating a new one as buffer migrating is expensive.
|
||||
*
|
||||
* TODO: In the future this will support scaling bucket percentages based on which ones filled first. This is currently disabled.
|
||||
*/
|
||||
template<bitmap_allocator_settings settings = fex_default_allocator_settings>
|
||||
class atomic_segmented_bitmap_allocator final {
|
||||
public:
|
||||
~atomic_segmented_bitmap_allocator() {
|
||||
deinit();
|
||||
}
|
||||
|
||||
// Initialize segmented allocator asking for a specific arena size.
|
||||
// Bitset tracking arena allocations will allocate an independent buffer independent of arena size.
|
||||
void init(size_t arena_size) {
|
||||
deinit();
|
||||
|
||||
// Align up to page.
|
||||
arena_size = FEXCore::AlignUpPowerOf2(arena_size, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
const auto total_bitset_tracking_memory = calculate_bucket_granules_and_bitset_sizes(arena_size);
|
||||
|
||||
// RWX for JIT buffer
|
||||
arena_base = reinterpret_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(arena_size, true));
|
||||
arena_base_size = arena_size;
|
||||
|
||||
// RW for bitset buffer
|
||||
const auto bitset_size_aligned_to_page = FEXCore::AlignUpPowerOf2(total_bitset_tracking_memory, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
bitset_base = reinterpret_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(bitset_size_aligned_to_page, false));
|
||||
bitset_base_size = bitset_size_aligned_to_page;
|
||||
|
||||
// Name the buffer. This is only going to be used for the JIT right now.
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", arena_base, arena_size);
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", bitset_base, bitset_base_size);
|
||||
|
||||
// THP enabled for both buffers.
|
||||
FEXCore::Allocator::VirtualTHPControl(arena_base, arena_size, FEXCore::Allocator::THPControl::Enable);
|
||||
FEXCore::Allocator::VirtualTHPControl(bitset_base, bitset_base_size, FEXCore::Allocator::THPControl::Enable);
|
||||
|
||||
initialize_bitsets_for_buckets();
|
||||
}
|
||||
|
||||
void deinit() {
|
||||
if (arena_base) {
|
||||
FEXCore::Allocator::VirtualFree(arena_base, arena_base_size);
|
||||
}
|
||||
|
||||
if (bitset_base) {
|
||||
FEXCore::Allocator::VirtualFree(bitset_base, bitset_base_size);
|
||||
}
|
||||
|
||||
arena_base = nullptr;
|
||||
bitset_base = nullptr;
|
||||
}
|
||||
|
||||
// Allocate memory.
|
||||
// Returns nullptr on failure to allocate.
|
||||
void* allocate(size_t size) {
|
||||
const auto allocation_order = find_allocation_order(size);
|
||||
for (size_t i = 0; i < num_bitset_buckets; ++i) {
|
||||
const auto bitset_index = get_bitset_index_from_allocation_order(allocation_order, i);
|
||||
auto& bucket = bitset_buckets[bitset_index];
|
||||
|
||||
if (!bucket.bucket_allocation_size) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto bucket_granule_size = bitset_buckets[bitset_index].bucket_granule_size;
|
||||
const auto aligned_size = FEXCore::AlignUpPowerOf2(size, bucket_granule_size);
|
||||
const auto bits_to_allocate = FEXCore::DividePow2(aligned_size, bucket_granule_size);
|
||||
|
||||
auto allocated_bitset_index = bucket.atomic_bitset.allocate(bits_to_allocate);
|
||||
|
||||
if (allocated_bitset_index == bucket.atomic_bitset.invalid()) {
|
||||
// Mark that the bucket might be full and continue going.
|
||||
mark_bucket_potentially_full(bucket, bitset_index);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Bitset range allocated, get the pointer.
|
||||
return get_ptr_from_bitset(bucket, allocated_bitset_index);
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// Free memory.
|
||||
void free(void* ptr, size_t size) {
|
||||
for (auto& bucket : bitset_buckets) {
|
||||
if (ptr >= bucket.bucket_allocation_base && ptr < (bucket.bucket_allocation_base + bucket.bucket_allocation_size)) {
|
||||
// Pointer base must be aligned to granule size, but size doesn't need to be.
|
||||
// User could have asked for a smaller size and we rounded up to the larger granule.
|
||||
LOGMAN_THROW_A_FMT(reinterpret_cast<uint64_t>(ptr) % bucket.bucket_granule_size == 0, "Pointer was not aligned to bucket granule "
|
||||
"size!");
|
||||
|
||||
const auto bitset_index = FEXCore::DividePow2(
|
||||
(reinterpret_cast<uint64_t>(ptr) - reinterpret_cast<uint64_t>(bucket.bucket_allocation_base)), bucket.bucket_granule_size);
|
||||
const auto bitset_count = FEXCore::DividePow2(FEXCore::AlignUpPowerOf2(size, bucket.bucket_granule_size), bucket.bucket_granule_size);
|
||||
bucket.atomic_bitset.free(bitset_index, bitset_count);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::AFmt("Attempted to free a pointer that isn't tracked in the bitmap?");
|
||||
}
|
||||
|
||||
// Completely clear the allocator.
|
||||
// Not thread safe!
|
||||
void clear() {
|
||||
// VirtualDontNeed replaces pages with zero page.
|
||||
FEXCore::Allocator::VirtualDontNeed(arena_base, arena_base_size);
|
||||
FEXCore::Allocator::VirtualDontNeed(bitset_base, bitset_base_size);
|
||||
|
||||
// Reinitialize the bitsets.
|
||||
initialize_bitsets_for_buckets();
|
||||
bucket_filled_order = ~0ULL;
|
||||
}
|
||||
|
||||
// Reevaluate bucket allocation capacity percentages.
|
||||
// Not thread safe!
|
||||
void reevaluate_bucket_allocations_percentages() {
|
||||
if (bucket_filled_order.load() == ~0ULL) {
|
||||
// Buckets never claimed that they could be full.
|
||||
return;
|
||||
}
|
||||
|
||||
// TODO: Adjust `bitset_allocation_percentages` based on fill order.
|
||||
// TODO: Reallocate bitsets with `calculate_bucket_granules_and_bitset_sizes`
|
||||
// Statistically for a Denuvo game, likely bucket[0] will be the first to fill, which will consume additional resources from the larger buckets.
|
||||
// Statistically for any other game, likely bucket[1] will be the first to fill, which will consume additional resources from bucket[0].
|
||||
// TODO: Gather growth statistics from a variety of titles to determine trends.
|
||||
|
||||
// Reset bucket fill order.
|
||||
bucket_filled_order = ~0ULL;
|
||||
}
|
||||
|
||||
// Debugger routines.
|
||||
// Returns a bucket index if the pointer exists in it.
|
||||
ssize_t find_bucket_index(void* ptr) const {
|
||||
for (size_t i = 0; i < bitset_buckets.size(); ++i) {
|
||||
const auto& bucket = bitset_buckets[i];
|
||||
if (ptr >= bucket.bucket_allocation_base && ptr < (bucket.bucket_allocation_base + bucket.bucket_allocation_size)) {
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Returns the number of buckets.
|
||||
// Although they may be empty without the ability to allocate.
|
||||
size_t num_buckets() const {
|
||||
return bitset_buckets.size();
|
||||
}
|
||||
|
||||
struct bucket_alloc_information {
|
||||
size_t granule_size {};
|
||||
size_t allocated {};
|
||||
size_t free {};
|
||||
};
|
||||
|
||||
// Returns information about a bucket's granule allocations.
|
||||
std::optional<bucket_alloc_information> get_granule_information(size_t bucket_index) const {
|
||||
if (bucket_index >= bitset_buckets.size()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const auto& bucket = bitset_buckets[bucket_index];
|
||||
const auto count = bucket.atomic_bitset.popcount();
|
||||
|
||||
return bucket_alloc_information {
|
||||
.granule_size = bucket.bucket_granule_size,
|
||||
.allocated = count,
|
||||
.free = bucket.atomic_bitset.size_in_bits() - count,
|
||||
};
|
||||
}
|
||||
|
||||
private:
|
||||
// Arena base.
|
||||
uint8_t* arena_base {};
|
||||
size_t arena_base_size {};
|
||||
|
||||
// Bitset tracking.
|
||||
uint8_t* bitset_base {};
|
||||
size_t bitset_base_size {};
|
||||
|
||||
// Bucket sizes are important for how much can be allocated in a single 64-bit atomic operation.
|
||||
// Ensure these two buffers are sorted by size.
|
||||
constexpr static std::array<uint32_t, 3> bucket_granule_sizes = settings.bucket_sizes;
|
||||
|
||||
// These must be power of two.
|
||||
static_assert(std::has_single_bit(bucket_granule_sizes[0]));
|
||||
static_assert(std::has_single_bit(bucket_granule_sizes[1]));
|
||||
static_assert(std::has_single_bit(bucket_granule_sizes[2]));
|
||||
|
||||
// Ordered by size.
|
||||
static_assert(bucket_granule_sizes[0] < bucket_granule_sizes[1]);
|
||||
static_assert(bucket_granule_sizes[1] < bucket_granule_sizes[2]);
|
||||
|
||||
constexpr static size_t num_bitset_buckets = bucket_granule_sizes.size();
|
||||
constexpr static size_t bitset_index_shift = 2;
|
||||
constexpr static size_t bitset_index_mask = (1U << bitset_index_shift) - 1;
|
||||
|
||||
struct bucket_information {
|
||||
// Memory that the bitset is controlling
|
||||
uint8_t* bucket_allocation_base {};
|
||||
size_t bucket_allocation_size {};
|
||||
|
||||
// Memory backing the bitset itself.
|
||||
uint8_t* bitset_memory_base {};
|
||||
size_t bitset_memory_size {};
|
||||
|
||||
// The allocation granule of the bitset.
|
||||
size_t bucket_granule_size {};
|
||||
|
||||
// std::atomic<bool>::fetch_or isn't required to be implemented, so use uint8_t instead.
|
||||
std::atomic<uint8_t> marked_potentially_full {};
|
||||
FEXCore::Utils::atomic_bitset<true, false> atomic_bitset {};
|
||||
};
|
||||
|
||||
// Order in which the buckets filled for heuristic scaling of bucket sizes.
|
||||
std::atomic<uint64_t> bucket_filled_order {~0ULL};
|
||||
std::array<bucket_information, num_bitset_buckets> bitset_buckets {};
|
||||
|
||||
// TODO: Support scaling bitset bucket percentages by order in which they filled.
|
||||
// Currently this is const and can't change when the buckets run out of space.
|
||||
constexpr static std::array<uint64_t, num_bitset_buckets> bitset_allocation_percentages {
|
||||
settings.bucket_allocation_percentages[0],
|
||||
settings.bucket_allocation_percentages[1],
|
||||
settings.bucket_allocation_percentages[2],
|
||||
};
|
||||
|
||||
// bucket[0] is special cased to catch everything remaining.
|
||||
static_assert(settings.bucket_allocation_percentages[0] == 100);
|
||||
|
||||
// Calculate bitset sizes based on percentages of the arena size.
|
||||
// eg - split at 50+40+10:
|
||||
// - 16MB: 8MB + 6.4MB + 1.6MB
|
||||
// - 128MB: 64MB + 51.2MB + 12.8MB
|
||||
size_t calculate_bucket_granules_and_bitset_sizes(size_t arena_size) {
|
||||
size_t total_bitset_range {};
|
||||
size_t total_bitset_tracking_memory {};
|
||||
|
||||
for (size_t inverse_i = bucket_granule_sizes.size(); inverse_i > 0; --inverse_i) {
|
||||
const auto i = inverse_i - 1;
|
||||
const size_t bucket_granule_size = bucket_granule_sizes[i];
|
||||
const auto allocation_percentage = bitset_allocation_percentages[i];
|
||||
auto& bucket = bitset_buckets[i];
|
||||
|
||||
// Set the bucket granule size.
|
||||
bucket.bucket_granule_size = bucket_granule_size;
|
||||
|
||||
// Calculate the amount of space this bitset tracks.
|
||||
// If each bucket is tracking data at all it must track at least one 64-bit atomic word of bits.
|
||||
// 16B bucket: 1KB minimum
|
||||
// 512B bucket: 32KB minimum
|
||||
// 2KB bucket: 128KB minimum
|
||||
// Total: 161KB to reach minimums.
|
||||
//
|
||||
// If minimum of each bucket isn't reached, that bucket allocates **ZERO**
|
||||
size_t bucket_track_size {};
|
||||
if (allocation_percentage == 100) {
|
||||
// Use the remaining size to allocate in special case.
|
||||
bucket_track_size = FEXCore::AlignDown(arena_size - total_bitset_range, bucket_granule_size * WORD_SIZE_BITS);
|
||||
} else {
|
||||
bucket_track_size = FEXCore::AlignDown(arena_size * allocation_percentage / 100, bucket_granule_size * WORD_SIZE_BITS);
|
||||
}
|
||||
|
||||
const auto tracked_bits = FEXCore::DividePow2(bucket_track_size, bucket_granule_size);
|
||||
const auto bitset_memory_size = tracked_bits / 8;
|
||||
bucket.bucket_allocation_size = bucket_track_size;
|
||||
bucket.bitset_memory_size = bitset_memory_size;
|
||||
|
||||
// Mark the bucket as empty if it has no size.
|
||||
bucket.marked_potentially_full.store(bucket_track_size == 0, std::memory_order_relaxed);
|
||||
total_bitset_range += bucket_track_size;
|
||||
total_bitset_tracking_memory += bitset_memory_size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(tracked_bits % WORD_SIZE_BITS == 0, "Wasn't aligned to 64-bits!");
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT((arena_size - total_bitset_range) == 0, "Still had {} bytes remaining in bitmap allocator init",
|
||||
(arena_size - total_bitset_range));
|
||||
|
||||
return total_bitset_tracking_memory;
|
||||
}
|
||||
|
||||
void initialize_bitsets_for_buckets() {
|
||||
uint8_t* current_base_arena = arena_base;
|
||||
uint8_t* current_base_bitset = bitset_base;
|
||||
for (auto& bucket : bitset_buckets) {
|
||||
if (!bucket.bucket_allocation_size) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto tracked_bits = FEXCore::DividePow2(bucket.bucket_allocation_size, bucket.bucket_granule_size);
|
||||
|
||||
bucket.bucket_allocation_base = current_base_arena;
|
||||
bucket.atomic_bitset.init(current_base_bitset, tracked_bits);
|
||||
|
||||
current_base_arena += bucket.bucket_allocation_size;
|
||||
current_base_bitset += bucket.bitset_memory_size;
|
||||
}
|
||||
|
||||
current_base_bitset =
|
||||
reinterpret_cast<uint8_t*>(FEXCore::AlignUpPowerOf2(reinterpret_cast<uint64_t>(current_base_bitset), FEXCore::Utils::FEX_PAGE_SIZE));
|
||||
LogMan::Throw::AFmt(current_base_arena == (arena_base + arena_base_size), "Failed arena math: 0x{:x} != expected 0x{:x}",
|
||||
(uint64_t)current_base_arena, (uint64_t)arena_base + arena_base_size);
|
||||
LogMan::Throw::AFmt(current_base_bitset == (bitset_base + bitset_base_size), "Failed bitset math: 0x{:x} != expected 0x{:x}",
|
||||
(uint64_t)current_base_bitset, (uint64_t)bitset_base + bitset_base_size);
|
||||
}
|
||||
|
||||
// Marking a bucket as potentially being full.
|
||||
// Doesn't necessarily mean it is actually full, just allocations have started failing.
|
||||
// Which this could mean high fragmentation or actually full.
|
||||
// Regardless mark the buffer based on order of allocations failing.
|
||||
void mark_bucket_potentially_full(bucket_information& bucket, size_t bitset_index) {
|
||||
if (bucket.marked_potentially_full.load(std::memory_order_relaxed)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// fetch_or is faster than CAS here.
|
||||
bool previous_full = bucket.marked_potentially_full.fetch_or(true);
|
||||
if (previous_full) {
|
||||
// Another thread already marking it as full. Minor race.
|
||||
return;
|
||||
}
|
||||
|
||||
uint64_t expected = bucket_filled_order.load(std::memory_order_relaxed);
|
||||
uint64_t desired;
|
||||
|
||||
do {
|
||||
desired = (expected << bitset_index_shift) | bitset_index;
|
||||
} while (!bucket_filled_order.compare_exchange_strong(expected, desired));
|
||||
}
|
||||
|
||||
// Return a pointer to the data backing the bitset based on allocation index.
|
||||
void* get_ptr_from_bitset(const bucket_information& bucket, size_t allocation_index) const {
|
||||
return bucket.bucket_allocation_base + bucket.bucket_granule_size * allocation_index;
|
||||
}
|
||||
|
||||
// Returns an ordered list of bitsets for which bitsets we should allocate from.
|
||||
// Always returns all the bitsets but fitment of the heuristic may change the order.
|
||||
// LSB is highest priority, MSB is lowest priority.
|
||||
uint64_t find_allocation_order(size_t size) const {
|
||||
// Align up by the size of the smallest bucket size
|
||||
size = FEXCore::AlignUpPowerOf2(size, bitset_buckets[0].bucket_granule_size);
|
||||
|
||||
struct tracking {
|
||||
int32_t index {};
|
||||
};
|
||||
|
||||
// This is effectively setup to be a linear scan std::deque, but without the slow overhead of std::deque.
|
||||
std::array<tracking, num_bitset_buckets> remaining_bitsets = {{
|
||||
{.index = 0},
|
||||
{.index = 1},
|
||||
{.index = 2},
|
||||
}};
|
||||
|
||||
// Check exact size match.
|
||||
int32_t exact_match_index = -1;
|
||||
int32_t smallest_single_atomic_index = -1;
|
||||
for (auto it = remaining_bitsets.begin(); it != remaining_bitsets.end(); ++it) {
|
||||
auto& bitset_tracking = *it;
|
||||
const auto bitset_index = bitset_tracking.index;
|
||||
const auto bucket_granule_size = bitset_buckets[bitset_index].bucket_granule_size;
|
||||
|
||||
if (bucket_granule_size == size) {
|
||||
exact_match_index = bitset_index;
|
||||
bitset_tracking.index = -1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for smallest match within a 64-bit atomic.
|
||||
for (auto it = remaining_bitsets.begin(); it != remaining_bitsets.end(); ++it) {
|
||||
auto& bitset_tracking = *it;
|
||||
if (bitset_tracking.index == -1) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto bitset_index = bitset_tracking.index;
|
||||
const auto bucket_granule_size = bitset_buckets[bitset_index].bucket_granule_size;
|
||||
|
||||
if (size < bucket_granule_size) {
|
||||
// If the allocation size is smaller than a single bit, then skip this.
|
||||
// Ensures we don't burn 2KB buckets with 128B allocations unnecessarily.
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto bucket_granule_size_per_atomic_word = bucket_granule_size * WORD_SIZE_BITS;
|
||||
|
||||
if (size <= bucket_granule_size_per_atomic_word) {
|
||||
// Fits within a single 64-bit atomic.
|
||||
smallest_single_atomic_index = bitset_index;
|
||||
bitset_tracking.index = -1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t allocation_order {};
|
||||
uint64_t current_shift {};
|
||||
|
||||
// Exact match has highest priority.
|
||||
if (exact_match_index != -1) {
|
||||
allocation_order = exact_match_index;
|
||||
current_shift += bitset_index_shift;
|
||||
}
|
||||
|
||||
// Smallest fit within a single atomic word has second priority.
|
||||
if (smallest_single_atomic_index != -1) {
|
||||
allocation_order |= (smallest_single_atomic_index << current_shift);
|
||||
current_shift += bitset_index_shift;
|
||||
}
|
||||
|
||||
// Remaining is sorted by smallest->largest bucket size;
|
||||
// TODO: Might lead to the bitset allocation heuristic to favor scaling the lower bucket sizes to a larger percentage.
|
||||
for (auto it : remaining_bitsets) {
|
||||
if (it.index == -1) {
|
||||
continue;
|
||||
}
|
||||
|
||||
allocation_order |= (it.index << current_shift);
|
||||
current_shift += bitset_index_shift;
|
||||
}
|
||||
|
||||
return allocation_order;
|
||||
}
|
||||
|
||||
constexpr static size_t get_bitset_index_from_allocation_order(uint64_t allocation_order, size_t index) {
|
||||
return (allocation_order >> (index * bitset_index_shift)) & bitset_index_mask;
|
||||
}
|
||||
|
||||
constexpr static size_t WORD_SIZE_BITS = sizeof(uint64_t) * 8;
|
||||
};
|
||||
} // namespace FEXCore::Utils
|
||||
@@ -36,6 +36,7 @@ namespace Handler {
|
||||
enum ConfigOption {
|
||||
#define OPT_BASE(type, group, enum, json, default) CONFIG_##enum,
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
CONFIG_MAX,
|
||||
};
|
||||
|
||||
#define ENUMDEFINES
|
||||
@@ -116,11 +117,13 @@ namespace detail {
|
||||
FEX_DEFAULT_VISIBILITY void SetDataDirectory(std::string_view Path, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY void SetConfigDirectory(const std::string_view Path, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY void SetConfigFileLocation(std::string_view Path, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY void SetCacheDirectory(const std::string_view Path);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetDataDirectory(bool Global = false);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigDirectory(bool Global);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigFileLocation(bool Global = false);
|
||||
FEX_DEFAULT_VISIBILITY fextl::string GetApplicationConfig(const std::string_view Program, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetCacheDirectory();
|
||||
|
||||
using LayerValue = std::variant< fextl::string, StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
|
||||
@@ -228,6 +231,8 @@ FEX_DEFAULT_VISIBILITY std::optional<T> GetConv(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<fextl::string*> Get(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY void Set(ConfigOption Option, std::string_view Data);
|
||||
FEX_DEFAULT_VISIBILITY void Erase(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY fextl::string SerializeForCache();
|
||||
FEX_DEFAULT_VISIBILITY bool CheckConfigMatches(std::string_view Config);
|
||||
|
||||
template<typename T>
|
||||
class FEX_DEFAULT_VISIBILITY Value {
|
||||
|
||||
@@ -113,15 +113,12 @@ public:
|
||||
* @brief Create a new thread object that doesn't inherit any state.
|
||||
* Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The thread state to inherit from if not nullptr.
|
||||
*
|
||||
* @return A new InternalThreadState object for using with a new guest thread.
|
||||
*/
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState = nullptr) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(const FEXCore::Core::CPUState* NewThreadState = nullptr) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
#ifndef _WIN32
|
||||
@@ -136,6 +133,8 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
|
||||
|
||||
virtual void InitDiskCache() = 0;
|
||||
|
||||
virtual AbstractCodeCache& GetCodeCache() = 0;
|
||||
virtual void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter>) = 0;
|
||||
virtual void FlushAndCloseCodeMap() = 0;
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "FEXCore/Core/CodeCache.h"
|
||||
#include "FEXCore/Core/Context.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "FEXCore/Utils/File.h"
|
||||
#include "FEXCore/Utils/WorkQueueThread.h"
|
||||
#include "FEXCore/fextl/memory.h"
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <stdint.h>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace DiskCache {
|
||||
|
||||
namespace MesaFOZ {
|
||||
|
||||
#define FOSSILIZE_BLOB_HASH_LENGTH 40 /* SHA1 hexadecimal string length */
|
||||
|
||||
struct __attribute__((packed)) foz_payload_key {
|
||||
uint8_t bytes[FOSSILIZE_BLOB_HASH_LENGTH];
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) foz_payload_header {
|
||||
uint32_t payload_size;
|
||||
uint32_t format;
|
||||
uint32_t crc;
|
||||
uint32_t uncompressed_size;
|
||||
};
|
||||
|
||||
} // namespace MesaFOZ
|
||||
|
||||
class IndexedDB;
|
||||
|
||||
struct IndexEntry {
|
||||
IndexedDB* DB;
|
||||
uint64_t Offset;
|
||||
uint32_t Size;
|
||||
uint32_t GuestSize;
|
||||
XXH128_hash_t GuestHash;
|
||||
fextl::vector<uint32_t> GuestExtents;
|
||||
};
|
||||
|
||||
struct IndexCacheHead {
|
||||
struct IndexEntry MainEntry;
|
||||
uint64_t MainEntryFootprint;
|
||||
fextl::unique_ptr<fextl::multimap<uint64_t, IndexEntry>> MoreEntries; // sorted by guest footprint
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) BlobFixedHeader {
|
||||
uint32_t GuestSize;
|
||||
uint32_t HostSize;
|
||||
uint32_t EntryPointCount;
|
||||
uint32_t SmallRelocCount;
|
||||
uint32_t ThunkRelocCount;
|
||||
XXH128_hash_t GuestHash;
|
||||
};
|
||||
|
||||
// packed struct for types 0, 2 and 3. type 1 is bigger and separate below
|
||||
struct __attribute__((packed)) BlobSmallRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t Type;
|
||||
union {
|
||||
struct __attribute__((packed)) {
|
||||
uint32_t Symbol;
|
||||
} Named;
|
||||
struct __attribute__((packed)) {
|
||||
uint64_t GuestRIP;
|
||||
} RIPLiteral;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint64_t GuestRIP;
|
||||
} RIPMove;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t ValueSize;
|
||||
uint32_t SiteOffset;
|
||||
} PatchableData;
|
||||
};
|
||||
};
|
||||
|
||||
// type 1, implicit
|
||||
struct __attribute__((packed)) BlobThunkRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t SymbolHash[32]; // sha256sum in the real RelocNamedThunkMove
|
||||
};
|
||||
|
||||
struct CodeHitData {
|
||||
fextl::vector<uint8_t> Blob;
|
||||
std::span<uint8_t> HostCode;
|
||||
std::span<const uint64_t> GuestPages;
|
||||
std::span<uint64_t> EntryPointRIPs;
|
||||
std::span<const uint32_t> EntryPointHostOffsets;
|
||||
|
||||
// the spans above point to memory owned by the Blob vec, so it's important this can't be copied
|
||||
CodeHitData() = default;
|
||||
CodeHitData(CodeHitData&&) = default;
|
||||
CodeHitData& operator=(CodeHitData&&) = default;
|
||||
CodeHitData(const CodeHitData&) = delete;
|
||||
CodeHitData& operator=(const CodeHitData&) = delete;
|
||||
};
|
||||
|
||||
using Index = fextl::robin_map<uint64_t, IndexCacheHead>;
|
||||
|
||||
class FOZFile {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheFileName, bool ReadOnly);
|
||||
bool Lock(uint32_t TimeoutMS) {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Lock(TimeoutMS);
|
||||
}
|
||||
bool Unlock() {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Unlock();
|
||||
}
|
||||
File::File::FileHandleType GetHandle() {
|
||||
return FD ? FD->GetHandle() : (File::File::FileHandleType)-1;
|
||||
}
|
||||
ssize_t Size();
|
||||
bool ReadAll(fextl::vector<uint8_t>& Out); // from first blob
|
||||
bool ReadBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool WriteBlob(const MesaFOZ::foz_payload_key& Key, std::span<const std::span<const uint8_t>> BlobChunks, uint64_t& OutBlobOffset);
|
||||
|
||||
private:
|
||||
static constexpr uint32_t OPEN_LOCK_TIMEOUT_MS = 100;
|
||||
|
||||
fextl::string FileName;
|
||||
fextl::unique_ptr<File::File> FD;
|
||||
bool ReadOnly = false;
|
||||
};
|
||||
|
||||
class IndexedDB {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
void PopulateIndex(Index& CacheIndex, bool& FoundMetadata);
|
||||
bool ReadCacheBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob, Index& CacheIndex,
|
||||
std::mutex& IndexMutex, std::span<const uint8_t> IndexBlob);
|
||||
|
||||
private:
|
||||
// stores run on the Writer, so returning quick isn't as important
|
||||
static constexpr uint32_t STORE_LOCK_TIMEOUT_MS = 1000;
|
||||
static constexpr uint64_t BIG_MAPPING_SIZE = 1ULL << 33;
|
||||
static constexpr uint32_t LOOKUP_KEY_MAX_BUCKET_DEPTH = 20;
|
||||
|
||||
FOZFile CacheFOZ;
|
||||
uint8_t* CacheFileMapping = nullptr;
|
||||
std::atomic<uint64_t> CacheFileSize;
|
||||
FOZFile IndexFOZ;
|
||||
bool ReadOnly = false;
|
||||
};
|
||||
|
||||
class DiskCache {
|
||||
public:
|
||||
void Init(FEXCore::Context::ContextImpl* CTX);
|
||||
|
||||
std::optional<CodeHitData> Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP,
|
||||
std::optional<uint64_t>& GuestCodeKey);
|
||||
void Validate(uint64_t GuestCodeKey, const CodeHitData& Hit, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::optional<ExecutableFileSectionInfo> Region);
|
||||
bool Store(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP, uint64_t GuestCodeKey,
|
||||
std::span<const uint8_t> GuestCode, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations, const Frontend::Decoder::DecodedBlockInformation* DecodedBlockInfo);
|
||||
|
||||
bool IsWritingDiskCache() const {
|
||||
return WritingDiskCache;
|
||||
}
|
||||
bool IsReadingDiskCache() const {
|
||||
return ReadingDiskCache;
|
||||
}
|
||||
bool IsValidating() const {
|
||||
return Validation;
|
||||
}
|
||||
|
||||
private:
|
||||
bool OpenCacheDB(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
uint64_t MakeLookupKey(Core::InternalThreadState* Thread, const uint64_t ModuleOffset, bool Writable, bool MonoBackpatcher);
|
||||
|
||||
bool ReadingDiskCache {};
|
||||
bool WritingDiskCache {};
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
XXH128_hash_t BucketHash;
|
||||
fextl::vector<fextl::unique_ptr<IndexedDB>> ROCacheDBs;
|
||||
fextl::unique_ptr<IndexedDB> RWCacheDB;
|
||||
Index Index;
|
||||
std::mutex IndexLock;
|
||||
bool FoundMetadata = false;
|
||||
struct CacheStoreWorkItem;
|
||||
|
||||
// the Writer holds references to all this stuff above and needs to be last
|
||||
fextl::unique_ptr<WorkQueueThread> Writer;
|
||||
|
||||
FEX_CONFIG_OPT(EnableDiskCache, DISKCACHE);
|
||||
FEX_CONFIG_OPT(Validation, DISKCACHEVALIDATION);
|
||||
FEX_CONFIG_OPT(MapDiskCacheFiles, DISKCACHEFILEMAPPING);
|
||||
FEX_CONFIG_OPT(RelocationFilter, DISKCACHERELOCATIONFILTER);
|
||||
FEX_CONFIG_OPT(AnonCaching, DISKCACHEANONCACHING);
|
||||
FEX_CONFIG_OPT(BasePathOverride, DISKCACHEPATH);
|
||||
FEX_CONFIG_OPT(RODBNames, DISKCACHERODBNAMES);
|
||||
};
|
||||
|
||||
static constexpr uint16_t AnonPrefixGuestBytes = 64;
|
||||
|
||||
// TODO: This header is in global installed header path, but uses internal headers.
|
||||
// Migrate this once that is fixed.
|
||||
static constexpr uint16_t FormatVersion = 20;
|
||||
FEX_DEFAULT_VISIBILITY uint16_t GetFormatVersion();
|
||||
|
||||
} // namespace DiskCache
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -0,0 +1,11 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
#include "FEXCore/Utils/File.h"
|
||||
|
||||
namespace FEXCore::DiskCache {
|
||||
using FileMapperFunc = void* (*)(FEXCore::File::File::FileHandleType Handle, uint64_t MapSize);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void SetFileMapper(FileMapperFunc Func);
|
||||
} // namespace FEXCore::DiskCache
|
||||
@@ -13,59 +13,71 @@ namespace FEXCore {
|
||||
* Not the x86->IR process
|
||||
*/
|
||||
struct HostFeatures {
|
||||
// Changes code generation slightly.
|
||||
enum class HostTypeEnum : uint32_t {
|
||||
Unknown,
|
||||
Linux,
|
||||
Wow64,
|
||||
Arm64ec,
|
||||
};
|
||||
|
||||
// Whether or not the host supports any kind of SVE implementation.
|
||||
[[nodiscard]]
|
||||
bool SupportsSVE() const {
|
||||
return SupportsSVE128 || SupportsSVE256;
|
||||
}
|
||||
|
||||
uint32_t DCacheLineSize {};
|
||||
uint32_t ICacheLineSize {};
|
||||
bool SupportsCacheMaintenanceOps {};
|
||||
bool SupportsAES {};
|
||||
bool SupportsCRC {};
|
||||
bool SupportsCLZERO {};
|
||||
bool SupportsAtomics {};
|
||||
bool SupportsRCPC {};
|
||||
bool SupportsTSOImm9 {};
|
||||
bool SupportsRAND {};
|
||||
bool SupportsAVX {};
|
||||
bool SupportsSVE128 {};
|
||||
bool SupportsSVE256 {};
|
||||
bool SupportsSHA {};
|
||||
bool SupportsPMULL_128Bit {};
|
||||
bool SupportsCSSC {};
|
||||
bool SupportsFCMA {};
|
||||
bool SupportsFlagM {};
|
||||
bool SupportsFlagM2 {};
|
||||
bool SupportsRPRES {};
|
||||
bool SupportsPreserveAllABI {};
|
||||
bool SupportsAES256 {};
|
||||
bool SupportsSVEBitPerm {};
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool SupportsFRINTTS {};
|
||||
bool SupportsECV {};
|
||||
bool SupportsWFXT {};
|
||||
bool Supports3DNow {};
|
||||
bool SupportsSSE4a {};
|
||||
bool SupportsMOPS {};
|
||||
bool PreferZVAForVZero {};
|
||||
[[nodiscard]]
|
||||
uint32_t DCacheSize() const {
|
||||
return 4 << DCacheLineLog2;
|
||||
}
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
bool SupportsFloatExceptions {};
|
||||
|
||||
// Changes code generation slightly.
|
||||
enum class HostTypeEnum {
|
||||
Unknown,
|
||||
Linux,
|
||||
Wow64,
|
||||
Arm64ec,
|
||||
};
|
||||
HostTypeEnum HostType {};
|
||||
[[nodiscard]]
|
||||
uint64_t HashForCaching() const {
|
||||
// As long as the number of options is 64-bit or below, we can just return it.
|
||||
// Skip CPUMIDRs as it doesn't affect codegen.
|
||||
static_assert(offsetof(HostFeatures, CPUMIDRs) == 8);
|
||||
uint64_t Result {};
|
||||
memcpy(&Result, this, sizeof(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint32_t DCacheLineLog2 : 4 {};
|
||||
uint32_t SupportsCacheMaintenanceOps : 1 {};
|
||||
uint32_t SupportsAES : 1 {};
|
||||
uint32_t SupportsCRC : 1 {};
|
||||
uint32_t SupportsCLZERO : 1 {};
|
||||
uint32_t SupportsAtomics : 1 {};
|
||||
uint32_t SupportsRCPC : 1 {};
|
||||
uint32_t SupportsTSOImm9 : 1 {};
|
||||
uint32_t SupportsRAND : 1 {};
|
||||
uint32_t SupportsAVX : 1 {};
|
||||
uint32_t SupportsSVE128 : 1 {};
|
||||
uint32_t SupportsSVE256 : 1 {};
|
||||
uint32_t SupportsSHA : 1 {};
|
||||
uint32_t SupportsPMULL_128Bit : 1 {};
|
||||
uint32_t SupportsCSSC : 1 {};
|
||||
uint32_t SupportsFCMA : 1 {};
|
||||
uint32_t SupportsFlagM : 1 {};
|
||||
uint32_t SupportsFlagM2 : 1 {};
|
||||
uint32_t SupportsRPRES : 1 {};
|
||||
uint32_t SupportsPreserveAllABI : 1 {};
|
||||
uint32_t SupportsAES256 : 1 {};
|
||||
uint32_t SupportsSVEBitPerm : 1 {};
|
||||
uint32_t SupportsCPUIndexInTPIDRRO : 1 {};
|
||||
uint32_t SupportsFRINTTS : 1 {};
|
||||
uint32_t SupportsECV : 1 {};
|
||||
uint32_t SupportsWFXT : 1 {};
|
||||
uint32_t Supports3DNow : 1 {};
|
||||
uint32_t SupportsSSE4a : 1 {};
|
||||
uint32_t SupportsMOPS : 1 {};
|
||||
uint32_t PreferZVAForVZero : 1 {};
|
||||
uint32_t SupportsAFP : 1 {};
|
||||
uint32_t SupportsFloatExceptions : 1 {};
|
||||
// Flag if this is InstCountCI
|
||||
bool IsInstCountCI {};
|
||||
uint32_t IsInstCountCI : 1 {};
|
||||
HostTypeEnum HostType : 2 {};
|
||||
uint32_t pad : 26 {};
|
||||
|
||||
// MIDR information
|
||||
// Also used for determining number of CPU cores for CPUID
|
||||
|
||||
@@ -17,29 +17,6 @@ struct CpuStateFrame;
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
struct SyscallArguments {
|
||||
static constexpr std::size_t MAX_ARGS = 7;
|
||||
uint64_t Argument[MAX_ARGS];
|
||||
};
|
||||
|
||||
struct SyscallABI {
|
||||
// Expectation is that the backend will be aware of how to modify the arguments based on numbering
|
||||
// Only GPRs expected
|
||||
uint8_t NumArgs;
|
||||
// If the syscall has a return then it should be stored in the ABI specific syscall register
|
||||
// Linux = RAX
|
||||
bool HasReturn;
|
||||
|
||||
int32_t HostSyscallNumber;
|
||||
};
|
||||
|
||||
enum class SyscallOSABI {
|
||||
OS_UNKNOWN,
|
||||
OS_LINUX64,
|
||||
OS_LINUX32,
|
||||
OS_GENERIC, // No JIT-side argument handling, spill/fill all regs.
|
||||
};
|
||||
|
||||
struct ExecutableRangeInfo {
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
@@ -53,11 +30,8 @@ class SyscallHandler {
|
||||
public:
|
||||
virtual ~SyscallHandler() = default;
|
||||
|
||||
virtual uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) = 0;
|
||||
virtual void HandleSyscall(FEXCore::Core::CpuStateFrame* Frame) = 0;
|
||||
|
||||
SyscallOSABI GetOSABI() const {
|
||||
return OSABI;
|
||||
}
|
||||
virtual void MarkGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {}
|
||||
virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {}
|
||||
virtual void MarkOvercommitRange(uint64_t Start, uint64_t Length) {}
|
||||
@@ -72,8 +46,5 @@ public:
|
||||
}
|
||||
|
||||
virtual void SleepThread(FEXCore::Context::Context* CTX, FEXCore::Core::CpuStateFrame* Frame) {}
|
||||
|
||||
protected:
|
||||
SyscallOSABI OSABI;
|
||||
};
|
||||
} // namespace FEXCore::HLE
|
||||
@@ -166,6 +166,8 @@ FEX_DEFAULT_VISIBILITY extern void InitializeThread();
|
||||
|
||||
#ifndef _WIN32
|
||||
void SetupAllocatorHooks(void* (*)(void* addr, size_t length, int prot, int flags, int fd, off_t offset), int (*)(void* addr, size_t length));
|
||||
#else
|
||||
void SetupAllocatorHooks(void (*)(const char* name, const void* address, size_t size));
|
||||
#endif
|
||||
|
||||
struct FEXAllocOperators {
|
||||
|
||||
@@ -3,10 +3,17 @@
|
||||
#include <FEXCore/fextl/allocator.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
|
||||
#include <chrono>
|
||||
#include <thread>
|
||||
#include <utility>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/file.h>
|
||||
#include <sys/stat.h>
|
||||
#else
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#include <windows.h>
|
||||
@@ -39,7 +46,8 @@ public:
|
||||
|
||||
File() = default;
|
||||
|
||||
File(const char* Filepath, FileModes Modes) {
|
||||
File(const char* Filepath, FileModes Modes, bool Seekable = true)
|
||||
: Seekable {Seekable} {
|
||||
#ifndef _WIN32
|
||||
auto Disp = TranslateModes(Modes);
|
||||
Handle = open(Filepath, Disp, DEFAULT_USER_PERMS);
|
||||
@@ -62,6 +70,38 @@ public:
|
||||
ShouldClose = IsValidHandle;
|
||||
}
|
||||
|
||||
File(File&& Other) noexcept
|
||||
: ShouldClose(std::exchange(Other.ShouldClose, false))
|
||||
, IsValidHandle(std::exchange(Other.IsValidHandle, false))
|
||||
#ifdef _WIN32
|
||||
, Handle(std::exchange(Other.Handle, INVALID_HANDLE_VALUE))
|
||||
#else
|
||||
, Handle(std::exchange(Other.Handle, -1))
|
||||
#endif
|
||||
, Seekable(std::exchange(Other.Seekable, false))
|
||||
, Locked(std::exchange(Other.Locked, false)) {
|
||||
}
|
||||
|
||||
File& operator=(File&& Other) noexcept {
|
||||
if (this == &Other) {
|
||||
return *this;
|
||||
}
|
||||
|
||||
std::swap(ShouldClose, Other.ShouldClose);
|
||||
std::swap(IsValidHandle, Other.IsValidHandle);
|
||||
std::swap(Handle, Other.Handle);
|
||||
std::swap(Seekable, Other.Seekable);
|
||||
std::swap(Locked, Other.Locked);
|
||||
return *this;
|
||||
}
|
||||
|
||||
File(const File&) = delete;
|
||||
File& operator=(const File&) = delete;
|
||||
|
||||
FileHandleType GetHandle() {
|
||||
return Handle;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Write Bytes to File
|
||||
*
|
||||
@@ -71,6 +111,10 @@ public:
|
||||
* @return The number of bytes actually written or -1 on error.
|
||||
*/
|
||||
ssize_t Write(const void* Buffer, size_t Bytes) {
|
||||
if (!Seekable) {
|
||||
LOGMAN_THROW_A_FMT(false, "Can't use non-positioned ops on a non-seekable file!");
|
||||
return -1;
|
||||
}
|
||||
#ifndef _WIN32
|
||||
return write(Handle, Buffer, Bytes);
|
||||
#else
|
||||
@@ -97,6 +141,10 @@ public:
|
||||
* @return The number of bytes read or -1 on error.
|
||||
*/
|
||||
ssize_t Read(void* Buffer, size_t Bytes) {
|
||||
if (!Seekable) {
|
||||
LOGMAN_THROW_A_FMT(false, "Can't use non-positioned ops on a non-seekable file!");
|
||||
return -1;
|
||||
}
|
||||
#ifndef _WIN32
|
||||
return read(Handle, Buffer, Bytes);
|
||||
#else
|
||||
@@ -110,6 +158,80 @@ public:
|
||||
#endif
|
||||
}
|
||||
|
||||
ssize_t PRead(void* Buffer, size_t Bytes, uint64_t Offset) {
|
||||
if (Seekable) {
|
||||
LOGMAN_THROW_A_FMT(false, "Can't use positioned ops on a seekable file!");
|
||||
return -1;
|
||||
}
|
||||
#ifndef _WIN32
|
||||
return pread(Handle, Buffer, Bytes, Offset);
|
||||
#else
|
||||
DWORD BytesRead {};
|
||||
OVERLAPPED Overlapped {};
|
||||
Overlapped.Offset = static_cast<DWORD>(Offset);
|
||||
Overlapped.OffsetHigh = static_cast<DWORD>(Offset >> 32);
|
||||
auto Result = ReadFile(Handle, Buffer, Bytes, &BytesRead, &Overlapped);
|
||||
if (Result) {
|
||||
return BytesRead;
|
||||
}
|
||||
// Some error, match Linux side.
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
|
||||
ssize_t PWrite(const void* Buffer, size_t Bytes, uint64_t Offset) {
|
||||
if (Seekable) {
|
||||
LOGMAN_THROW_A_FMT(false, "Can't use positioned ops on a seekable file!");
|
||||
return -1;
|
||||
}
|
||||
#ifndef _WIN32
|
||||
return pwrite(Handle, Buffer, Bytes, Offset);
|
||||
#else
|
||||
DWORD BytesWritten {};
|
||||
OVERLAPPED Overlapped {};
|
||||
Overlapped.Offset = static_cast<DWORD>(Offset);
|
||||
Overlapped.OffsetHigh = static_cast<DWORD>(Offset >> 32);
|
||||
auto Result = WriteFile(Handle, Buffer, Bytes, &BytesWritten, &Overlapped);
|
||||
if (Result) {
|
||||
return BytesWritten;
|
||||
}
|
||||
// Some error, match Linux side.
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
|
||||
bool Lock(uint32_t TimeoutMS) {
|
||||
for (uint32_t i = 0;; ++i) {
|
||||
if (TryLock()) {
|
||||
return true;
|
||||
}
|
||||
if (i >= TimeoutMS) {
|
||||
return false;
|
||||
}
|
||||
std::this_thread::sleep_for(std::chrono::milliseconds(1));
|
||||
}
|
||||
}
|
||||
|
||||
bool Unlock() {
|
||||
if (!Locked) {
|
||||
return false; // we could return true here :thonk:
|
||||
}
|
||||
#ifndef _WIN32
|
||||
if (flock(Handle, LOCK_UN) == -1) {
|
||||
return false;
|
||||
}
|
||||
#else
|
||||
OVERLAPPED Overlapped {};
|
||||
Overlapped.Offset = static_cast<DWORD>(LOCK_SENTINEL_OFFSET);
|
||||
Overlapped.OffsetHigh = static_cast<DWORD>(LOCK_SENTINEL_OFFSET >> 32);
|
||||
if (!UnlockFileEx(Handle, 0, 1, 0, &Overlapped)) {
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
Locked = false;
|
||||
return true;
|
||||
}
|
||||
|
||||
~File() {
|
||||
if (!IsValidHandle) {
|
||||
return;
|
||||
@@ -166,6 +288,22 @@ public:
|
||||
#endif
|
||||
}
|
||||
|
||||
ssize_t Size() {
|
||||
#ifndef _WIN32
|
||||
struct stat st;
|
||||
if (fstat(Handle, &st) != 0) {
|
||||
return -1;
|
||||
}
|
||||
return st.st_size;
|
||||
#else
|
||||
LARGE_INTEGER FileSize;
|
||||
if (!GetFileSizeEx(Handle, &FileSize)) {
|
||||
return -1;
|
||||
}
|
||||
return FileSize.QuadPart;
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Seek the file pointer location.
|
||||
*
|
||||
@@ -175,6 +313,10 @@ public:
|
||||
* @return The current file pointer location or -1.
|
||||
*/
|
||||
ssize_t Seek(ssize_t Distance, SeekOp Op) {
|
||||
if (!Seekable) {
|
||||
LOGMAN_THROW_A_FMT(false, "Can't use non-positioned ops on a non-seekable file!");
|
||||
return -1;
|
||||
}
|
||||
#ifndef _WIN32
|
||||
return lseek(Handle, Distance, TranslateSeek(Op));
|
||||
#else
|
||||
@@ -191,15 +333,39 @@ public:
|
||||
|
||||
protected:
|
||||
|
||||
File(FileHandleType Handle, bool ShouldClose)
|
||||
File(FileHandleType Handle, bool ShouldClose, bool Seekable = true)
|
||||
: ShouldClose {ShouldClose}
|
||||
, IsValidHandle {true}
|
||||
, Handle {Handle} {}
|
||||
, Handle {Handle}
|
||||
, Seekable {Seekable} {}
|
||||
private:
|
||||
bool TryLock() {
|
||||
if (Locked) {
|
||||
return true;
|
||||
}
|
||||
#ifndef _WIN32
|
||||
if (flock(Handle, LOCK_EX | LOCK_NB) == -1) {
|
||||
return false;
|
||||
}
|
||||
#else
|
||||
// mimic posix advisory-only lock by locking some unattainably-high bit
|
||||
OVERLAPPED Overlapped {};
|
||||
Overlapped.Offset = static_cast<DWORD>(LOCK_SENTINEL_OFFSET);
|
||||
Overlapped.OffsetHigh = static_cast<DWORD>(LOCK_SENTINEL_OFFSET >> 32);
|
||||
if (!LockFileEx(Handle, LOCKFILE_EXCLUSIVE_LOCK | LOCKFILE_FAIL_IMMEDIATELY, 0, 1, 0, &Overlapped)) {
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
Locked = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ShouldClose {};
|
||||
bool IsValidHandle {};
|
||||
|
||||
FileHandleType Handle {};
|
||||
bool Seekable = true;
|
||||
bool Locked = false;
|
||||
#ifndef _WIN32
|
||||
static constexpr int DEFAULT_USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
|
||||
@@ -240,6 +406,8 @@ private:
|
||||
}
|
||||
#else
|
||||
static constexpr int DEFAULT_SHARE_MODE = FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE;
|
||||
static constexpr uint64_t LOCK_SENTINEL_OFFSET = 1ULL << 62;
|
||||
|
||||
struct Disposition {
|
||||
uint32_t CreationFlag;
|
||||
uint32_t Access;
|
||||
@@ -254,7 +422,11 @@ private:
|
||||
Disp.Access |= GENERIC_WRITE;
|
||||
}
|
||||
if ((Modes & FileModes::CREATE) == FileModes::CREATE) {
|
||||
Disp.CreationFlag = CREATE_ALWAYS;
|
||||
if ((Modes & FileModes::TRUNCATE) == FileModes::TRUNCATE) {
|
||||
Disp.CreationFlag = CREATE_ALWAYS;
|
||||
} else {
|
||||
Disp.CreationFlag = OPEN_ALWAYS;
|
||||
}
|
||||
} else {
|
||||
Disp.CreationFlag = OPEN_ALWAYS;
|
||||
}
|
||||
|
||||
@@ -18,6 +18,18 @@ constexpr uint64_t AlignDown(uint64_t value, uint64_t size) {
|
||||
return value - value % size;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint64_t AlignUpPowerOf2(uint64_t value, uint64_t size) {
|
||||
LOGMAN_THROW_A_FMT(std::popcount(size) == 1, "Alignment needs to be power of 2");
|
||||
return (value + size - 1) & ~(size - 1);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint64_t AlignDownPowerOf2(uint64_t value, uint64_t size) {
|
||||
LOGMAN_THROW_A_FMT(std::popcount(size) == 1, "Alignment needs to be power of 2");
|
||||
return value & ~(size - 1);
|
||||
}
|
||||
|
||||
// Returns the ilog2 of a power-of-2 integer.
|
||||
// Asserts in the case that the passed in integer is not a power-of-2.
|
||||
template<typename T>
|
||||
|
||||
@@ -70,6 +70,11 @@ struct ThreadStats {
|
||||
uint64_t AccumulatedCacheWriteLockTime;
|
||||
|
||||
uint64_t AccumulatedJITCount;
|
||||
|
||||
uint64_t AccumulatedDiskCacheHitCount;
|
||||
uint64_t AccumulatedDiskCacheMissCount;
|
||||
uint64_t AccumulatedDiskCacheLookupTime;
|
||||
uint64_t Padding;
|
||||
};
|
||||
|
||||
// Ensure 16-byte alignment to take advantage of ARM single-copy atomicity.
|
||||
|
||||
@@ -250,19 +250,19 @@ public:
|
||||
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]]
|
||||
static auto MaskSignalsAndLockMutex(MutexType& mutex, uint64_t Mask = ~0ULL) {
|
||||
static inline auto MaskSignalsAndLockMutex(MutexType& mutex, uint64_t Mask = ~0ULL) {
|
||||
return LockType<MutexType> {mutex};
|
||||
}
|
||||
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]]
|
||||
static auto GuardSignalDeferringSection(MutexType& mutex, FEXCore::Core::InternalThreadState* Thread, uint64_t Mask = ~0ULL) {
|
||||
static inline auto GuardSignalDeferringSection(MutexType& mutex, FEXCore::Core::InternalThreadState* Thread, uint64_t Mask = ~0ULL) {
|
||||
return LockType<MutexType> {mutex};
|
||||
}
|
||||
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]]
|
||||
static auto GuardSignalDeferringSectionWithFallback(MutexType& mutex, FEXCore::Core::InternalThreadState* Thread, uint64_t Mask = ~0ULL) {
|
||||
static inline auto GuardSignalDeferringSectionWithFallback(MutexType& mutex, FEXCore::Core::InternalThreadState* Thread, uint64_t Mask = ~0ULL) {
|
||||
return LockType<MutexType> {mutex};
|
||||
}
|
||||
|
||||
|
||||
@@ -162,6 +162,9 @@ public:
|
||||
std::optional<ContainerType::iterator> TryToReownBuffer(const ContainerType::iterator& Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
ClientFlags Expected = ClientFlags::FLAG_DISOWNED;
|
||||
if (!CurrentClientFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_OWNED)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(Expected != IntrusivePooledAllocator::ClientFlags::FLAG_OWNED, "Tried to reown buffer but it was already owned!");
|
||||
#endif
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
@@ -540,6 +543,20 @@ public:
|
||||
return ReownOrClaimBufferWithSize(NewSize).Ptr;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Assertion feature to check if current buffer is in a disowned or free state.
|
||||
*
|
||||
* Useful to validate that buffers are disowned at the correct places in code.
|
||||
*/
|
||||
void ValidateDisownedOrFree() const {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto OwnedState = ClientOwnedFlag.load();
|
||||
LOGMAN_THROW_A_FMT(
|
||||
OwnedState == IntrusivePooledAllocator::ClientFlags::FLAG_DISOWNED || OwnedState == IntrusivePooledAllocator::ClientFlags::FLAG_FREE,
|
||||
"ThreadPoolAllocator should have been disowned!");
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Disown or unclaim the buffer, letting the `Allocator` know it can reclaim the buffer
|
||||
*
|
||||
|
||||
@@ -4,10 +4,16 @@
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore::Threads {
|
||||
|
||||
struct Flags {
|
||||
bool LowPriority : 1 {};
|
||||
bool Internal : 1 {};
|
||||
};
|
||||
|
||||
using ThreadFunc = void* (*)(void* user_ptr);
|
||||
|
||||
class Thread;
|
||||
using CreateThreadFunc = fextl::unique_ptr<Thread> (*)(ThreadFunc Func, void* Arg);
|
||||
using CreateThreadFunc = fextl::unique_ptr<Thread> (*)(ThreadFunc Func, void* Arg, Flags Flags);
|
||||
using CleanupAfterForkFunc = void (*)();
|
||||
|
||||
struct Pointers {
|
||||
@@ -28,7 +34,7 @@ public:
|
||||
* @name Calls provided API functions
|
||||
* @{ */
|
||||
|
||||
static fextl::unique_ptr<Thread> Create(ThreadFunc Func, void* Arg);
|
||||
static fextl::unique_ptr<Thread> Create(ThreadFunc Func, void* Arg, Flags Flags = {});
|
||||
|
||||
static void CleanupAfterFork();
|
||||
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <condition_variable>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
class WorkQueueThread {
|
||||
public:
|
||||
// destroyed after Run()
|
||||
struct WorkItem {
|
||||
virtual ~WorkItem() = default;
|
||||
virtual void Run() = 0;
|
||||
};
|
||||
|
||||
WorkQueueThread(FEXCore::Threads::Flags Flags = {});
|
||||
~WorkQueueThread();
|
||||
|
||||
void QueueWork(fextl::unique_ptr<WorkItem> Work);
|
||||
|
||||
private:
|
||||
// static function for the ::Thread to refer to
|
||||
static void* ThreadEntry(void* Self) {
|
||||
static_cast<WorkQueueThread*>(Self)->ThreadProc();
|
||||
return nullptr;
|
||||
}
|
||||
void ThreadProc();
|
||||
|
||||
std::mutex Mutex;
|
||||
std::condition_variable CV;
|
||||
fextl::deque<fextl::unique_ptr<WorkItem>> Queue;
|
||||
bool Stop = false;
|
||||
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> Thread;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -2,6 +2,7 @@
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <limits.h>
|
||||
|
||||
#if !defined(_WIN32)
|
||||
#include <linux/futex.h> /* Definition of FUTEX_* constants */
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_all.hpp>
|
||||
#include <unordered_set>
|
||||
|
||||
#include "Utils/atomic_segmented_bitmap_allocator.h"
|
||||
|
||||
TEST_CASE("Single") {
|
||||
FEXCore::Utils::atomic_segmented_bitmap_allocator alloc {};
|
||||
alloc.init(4096);
|
||||
alloc.init(162 * 1024);
|
||||
alloc.init(1280 * 1024);
|
||||
}
|
||||
|
||||
TEST_CASE("Single - tiny") {
|
||||
FEXCore::Utils::atomic_segmented_bitmap_allocator alloc {};
|
||||
alloc.init(4096);
|
||||
std::map<void*, uint64_t> hits {};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto ptr = alloc.allocate(16);
|
||||
CHECK(ptr != nullptr);
|
||||
|
||||
// Count any potential duplicate hits.
|
||||
hits[ptr] += 1;
|
||||
}
|
||||
|
||||
// Ensure all hits aren't duplicated.
|
||||
for (auto hit : hits) {
|
||||
CHECK(hit.second == 1);
|
||||
}
|
||||
|
||||
// Final allocation should fail.
|
||||
CHECK(alloc.allocate(16) == nullptr);
|
||||
}
|
||||
|
||||
TEST_CASE("Single - tiny failover") {
|
||||
FEXCore::Utils::atomic_segmented_bitmap_allocator alloc {};
|
||||
|
||||
// Should give 6400 + 128 + 0 allocation counts with the default buckets.
|
||||
const size_t allocation_size = 164 * 1024;
|
||||
alloc.init(allocation_size);
|
||||
|
||||
std::map<void*, uint64_t> hits {};
|
||||
std::unordered_set<ssize_t> hit_buckets;
|
||||
|
||||
void* ptr {};
|
||||
while ((ptr = alloc.allocate(16)) != nullptr) {
|
||||
// Count any potential duplicate hits.
|
||||
hits[ptr] += 1;
|
||||
hit_buckets.emplace(alloc.find_bucket_index(ptr));
|
||||
}
|
||||
|
||||
// Ensure all hits aren't duplicated.
|
||||
for (auto hit : hits) {
|
||||
CHECK(hit.second == 1);
|
||||
}
|
||||
|
||||
// Ensure at least two buckets were hit.
|
||||
size_t buckets {};
|
||||
size_t total_memory {};
|
||||
for (auto index : hit_buckets) {
|
||||
REQUIRE(index != -1);
|
||||
buckets++;
|
||||
|
||||
const auto granule_information = alloc.get_granule_information(index);
|
||||
REQUIRE(granule_information.has_value());
|
||||
CHECK(granule_information->free == 0);
|
||||
total_memory += granule_information->allocated * granule_information->granule_size;
|
||||
}
|
||||
|
||||
CHECK(buckets > 1);
|
||||
CHECK(total_memory == allocation_size);
|
||||
|
||||
// Walk all the allocations and free them.
|
||||
for (auto hit : hits) {
|
||||
alloc.free(hit.first, 16);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Single - tiny - free") {
|
||||
FEXCore::Utils::atomic_segmented_bitmap_allocator alloc {};
|
||||
|
||||
std::vector<void*> values {};
|
||||
values.resize(256);
|
||||
alloc.init(4096);
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto ptr = alloc.allocate(16);
|
||||
values[i] = ptr;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
alloc.free(values[i], 16);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,657 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_all.hpp>
|
||||
#include <thread>
|
||||
|
||||
#include "Utils/atomic_bitset.h"
|
||||
|
||||
bool CheckMemoryIsZero(void* ptr, size_t size) {
|
||||
REQUIRE(size % sizeof(uint64_t) == 0);
|
||||
|
||||
auto ptr_u64 = reinterpret_cast<uint64_t*>(ptr);
|
||||
for (size_t i = 0; i < (size / sizeof(uint64_t)); ++i) {
|
||||
if (ptr_u64[i] != 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CheckMemoryIsSet(void* ptr, size_t size) {
|
||||
REQUIRE(size % sizeof(uint64_t) == 0);
|
||||
|
||||
auto ptr_u64 = reinterpret_cast<uint64_t*>(ptr);
|
||||
for (size_t i = 0; i < (size / sizeof(uint64_t)); ++i) {
|
||||
if (ptr_u64[i] != ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
struct buffer {
|
||||
uint8_t* ptr;
|
||||
|
||||
uint8_t* ptr_base;
|
||||
size_t size;
|
||||
};
|
||||
|
||||
buffer AllocateProtectedBuffer(size_t size) {
|
||||
buffer buf {
|
||||
.size = size + 4096 * 2,
|
||||
};
|
||||
buf.ptr_base = reinterpret_cast<uint8_t*>(mmap(nullptr, buf.size, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
REQUIRE(buf.ptr_base != nullptr);
|
||||
|
||||
buf.ptr = buf.ptr_base + 4096;
|
||||
|
||||
// RW only the pages requested.
|
||||
mprotect(buf.ptr, size, PROT_READ | PROT_WRITE);
|
||||
|
||||
return buf;
|
||||
}
|
||||
|
||||
void FreeProtectedBuffer(buffer buf) {
|
||||
munmap(buf.ptr_base, buf.size);
|
||||
}
|
||||
|
||||
TEST_CASE("Single") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Basic allocation check.
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
set.free(slot, 1);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("All") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate all bits, ensuring all can be allocated.
|
||||
for (size_t i = 0; i < size_bits; ++i) {
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
}
|
||||
|
||||
CHECK(CheckMemoryIsSet(buf.ptr, size));
|
||||
|
||||
// Ensure that overallocation fails.
|
||||
CHECK(set.allocate(1) == set.invalid());
|
||||
|
||||
// Free all the bits
|
||||
for (size_t i = 0; i < size_bits; ++i) {
|
||||
set.free(i, 1);
|
||||
}
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single 64-bit word.
|
||||
auto slot = set.allocate(64);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
set.free(slot, 64);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large Sparse") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single bit.
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
// Allocate a single 64-bit contiguous region.
|
||||
// Due to implementation behaviour, this should be a full word ahead of the previous.
|
||||
auto slot64 = set.allocate(64);
|
||||
REQUIRE(slot64 != set.invalid());
|
||||
CHECK(slot64 == 64);
|
||||
|
||||
set.free(slot, 1);
|
||||
set.free(slot64, 64);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large Sparse - in-fill") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single bit.
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
// Allocate a single 64-bit contiguous region.
|
||||
// Due to implementation behaviour, this should be a full word ahead of the previous.
|
||||
auto slot64 = set.allocate(64);
|
||||
REQUIRE(slot64 != set.invalid());
|
||||
CHECK(slot64 == 64);
|
||||
|
||||
std::vector<size_t> sparse {};
|
||||
|
||||
for (size_t i = 1; i < 64; ++i) {
|
||||
// Allocation of single elements should fill in sparsity.
|
||||
auto new_slot = set.allocate(1);
|
||||
REQUIRE(new_slot != set.invalid());
|
||||
CHECK(new_slot == i);
|
||||
sparse.emplace_back(new_slot);
|
||||
}
|
||||
|
||||
for (auto it : sparse) {
|
||||
set.free(it, 1);
|
||||
}
|
||||
|
||||
set.free(slot, 1);
|
||||
set.free(slot64, 64);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large Sparse - chunk") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single bit.
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
// Allocate a single 64-bit contiguous region.
|
||||
// Due to implementation behaviour, this should be a full word ahead of the previous.
|
||||
auto slot64 = set.allocate(64);
|
||||
REQUIRE(slot64 != set.invalid());
|
||||
CHECK(slot64 == 64);
|
||||
|
||||
// A smaller allocation that fits within an empty word should still sub allocate.
|
||||
auto slot32 = set.allocate(32);
|
||||
REQUIRE(slot32 != set.invalid());
|
||||
CHECK(slot32 < slot64);
|
||||
|
||||
set.free(slot, 1);
|
||||
set.free(slot64, 64);
|
||||
set.free(slot32, 32);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large Sparse - chunk in-fill") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single bit.
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
// Allocate a single 64-bit contiguous region.
|
||||
// Due to implementation behaviour, this should be a full word ahead of the previous.
|
||||
auto slot64 = set.allocate(64);
|
||||
REQUIRE(slot64 != set.invalid());
|
||||
CHECK(slot64 == 64);
|
||||
|
||||
// 63-bits should in-fill between the previous two allocations
|
||||
auto slot63 = set.allocate(63);
|
||||
REQUIRE(slot63 != set.invalid());
|
||||
CHECK(slot63 < slot64);
|
||||
CHECK(slot63 == (slot + 1));
|
||||
|
||||
set.free(slot, 1);
|
||||
set.free(slot64, 64);
|
||||
set.free(slot63, 63);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large to small") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single word.
|
||||
auto slot = set.allocate(64);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == 0);
|
||||
|
||||
// Clearing the sub bits individually should work.
|
||||
for (size_t i = 0; i < 64; ++i) {
|
||||
set.free(slot + i, 1);
|
||||
}
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Small to large") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate a single word using single allocations.
|
||||
auto slot0 = set.allocate(1);
|
||||
REQUIRE(slot0 != set.invalid());
|
||||
CHECK(slot0 == 0);
|
||||
|
||||
for (size_t i = 1; i < 64; ++i) {
|
||||
auto slot = set.allocate(1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
CHECK(slot == i);
|
||||
}
|
||||
|
||||
// Freeing smaller continguous slots using a larger size should work.
|
||||
set.free(slot0, 64);
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Large Clear") {
|
||||
constexpr size_t size = 4096 * 4;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
// Allocate the whole set
|
||||
while (set.allocate(64) != set.invalid())
|
||||
;
|
||||
|
||||
CHECK(CheckMemoryIsSet(buf.ptr, size));
|
||||
|
||||
// Clearing the set should reset everything.
|
||||
set.clear();
|
||||
|
||||
CHECK(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Larger than word") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
struct test_data {
|
||||
uint64_t offset_count {};
|
||||
uint64_t total_size {};
|
||||
};
|
||||
|
||||
auto run_test = [&set](test_data data) {
|
||||
if (data.offset_count) {
|
||||
REQUIRE(set.allocate(data.offset_count) == 0);
|
||||
}
|
||||
|
||||
REQUIRE(set.allocate(data.total_size) == data.offset_count);
|
||||
|
||||
for (size_t i = 0; i < data.offset_count; ++i) {
|
||||
REQUIRE(set.is_set(i));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < data.total_size; ++i) {
|
||||
REQUIRE(set.is_set(data.offset_count + i));
|
||||
}
|
||||
|
||||
// Reset the buffer.
|
||||
set.clear();
|
||||
};
|
||||
|
||||
// Test matrix to hit all the code paths for larger than word allocations
|
||||
//
|
||||
// | Head offset | Head | Head+Center | Head+Center+Tail |
|
||||
// | ----------- | ---- | ----------- | ---------------- |
|
||||
// | Offset(0) | 🗹 | 🗹 | 🗹 |
|
||||
// | Offset(1) | 🗹 | 🗹 | 🗹 |
|
||||
// | Offset(63) | 🗹 | 🗹 | 🗹 |
|
||||
// | Offset(511) | 🗹 | 🗹 | 🗹 |
|
||||
|
||||
constexpr static test_data tests[] = {
|
||||
{0, 64 + 1}, // Offset(0) + Head + Tail
|
||||
{1, 63 + 2}, // Offset(1) + Head + Tail
|
||||
{62, 2 + 63}, // Offset(62) + Head + Tail
|
||||
{510, 2 + 63}, // Offset(510) + Head + Tail
|
||||
{0, 64 + 64}, // Offset(0) + Head + Center
|
||||
{1, 63 + 64}, // Offset(1) + Head + Center
|
||||
{63, 1 + 64}, // Offset(63) + Head + Center
|
||||
{511, 1 + 64}, // Offset(511) + Head + Center
|
||||
{0, 64 + 64 + 1}, // Offset(0) + Head + Center + Tail
|
||||
{1, 63 + 64 + 1}, // Offset(1) + Head + Center + Tail
|
||||
{63, 1 + 64 + 1}, // Offset(63) + Head + Center + Tail
|
||||
{511, 1 + 64 + 1}, // Offset(511) + Head + Center + Tail
|
||||
};
|
||||
|
||||
for (auto test : tests) {
|
||||
run_test(test);
|
||||
}
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Larger than word - Free") {
|
||||
constexpr size_t size = 4096;
|
||||
constexpr size_t size_bits = size * 8;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset set {};
|
||||
set.init(buf.ptr, size_bits);
|
||||
|
||||
REQUIRE(set.size_in_bits() == size_bits);
|
||||
|
||||
struct test_data {
|
||||
uint64_t offset_count {};
|
||||
uint64_t total_size {};
|
||||
};
|
||||
|
||||
auto run_test = [&set, &buf](test_data data) {
|
||||
if (data.offset_count) {
|
||||
REQUIRE(set.allocate(data.offset_count) == 0);
|
||||
}
|
||||
|
||||
REQUIRE(set.allocate(data.total_size) == data.offset_count);
|
||||
|
||||
for (size_t i = 0; i < data.offset_count; ++i) {
|
||||
REQUIRE(set.is_set(i));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < data.total_size; ++i) {
|
||||
REQUIRE(set.is_set(data.offset_count + i));
|
||||
}
|
||||
|
||||
if (data.offset_count) {
|
||||
set.free(0, data.offset_count);
|
||||
}
|
||||
|
||||
set.free(data.offset_count, data.total_size);
|
||||
|
||||
REQUIRE(CheckMemoryIsZero(buf.ptr, size));
|
||||
|
||||
// Reset the buffer.
|
||||
set.clear();
|
||||
};
|
||||
|
||||
// Test matrix to hit all the code paths for larger than word allocations
|
||||
//
|
||||
// | Head offset | Head | Head+Center | Head+Center+Tail |
|
||||
// | ----------- | ---- | ----------- | ---------------- |
|
||||
// | Offset(0) | 🗹 | 🗹 | 🗹 |
|
||||
// | Offset(1) | 🗹 | 🗹 | 🗹 |
|
||||
// | Offset(63) | 🗹 | 🗹 | 🗹 |
|
||||
constexpr static test_data tests[] = {
|
||||
{0, 64 + 1}, // Offset(0) + Head + Tail
|
||||
{1, 63 + 2}, // Offset(1) + Head + Tail
|
||||
{62, 2 + 63}, // Offset(62) + Head + Tail
|
||||
{0, 64 + 64}, // Offset(0) + Head + Center
|
||||
{1, 63 + 64}, // Offset(1) + Head + Center
|
||||
{63, 1 + 64}, // Offset(63) + Head + Center
|
||||
{0, 64 + 64 + 1}, // Offset(0) + Head + Center + Tail
|
||||
{1, 63 + 64 + 1}, // Offset(1) + Head + Center + Tail
|
||||
{63, 1 + 64 + 1}, // Offset(63) + Head + Center + Tail
|
||||
};
|
||||
|
||||
for (auto test : tests) {
|
||||
run_test(test);
|
||||
}
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Larger than Word - 128-bits") {
|
||||
// Test to ensure on small bitset size, a larger than word allocation still fits.
|
||||
constexpr size_t size = 4096;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset<false, false> set {};
|
||||
|
||||
// Move the set up to the edge of the page to detect overruns.
|
||||
set.init(reinterpret_cast<uint8_t*>(buf.ptr) + (4096 - 16), 128);
|
||||
|
||||
REQUIRE(set.size_in_bits() == 128);
|
||||
|
||||
for (size_t i = 0; i < 128; ++i) {
|
||||
auto slot = set.allocate(i + 1);
|
||||
REQUIRE(slot != set.invalid());
|
||||
set.free(slot, i + 1);
|
||||
}
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Race acquire - Single") {
|
||||
// Test to ensure that allocating 1 slot should never fail unless it is actually full.
|
||||
// Basic race condition check.
|
||||
constexpr size_t size = 4096;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset<false, false> set {};
|
||||
|
||||
// Move the set up to the edge of the page to detect overruns.
|
||||
set.init(reinterpret_cast<uint8_t*>(buf.ptr) + (4096 - 32), 256);
|
||||
|
||||
REQUIRE(set.size_in_bits() == 256);
|
||||
|
||||
// Allocate all 256-bits
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
REQUIRE(set.allocate(1) != set.invalid());
|
||||
}
|
||||
|
||||
// Free the first 128-bits
|
||||
for (size_t i = 0; i < 128; ++i) {
|
||||
set.free(i, 1);
|
||||
}
|
||||
|
||||
std::atomic<bool> Acquire {};
|
||||
std::atomic<uint64_t> Waiters {};
|
||||
std::vector<std::thread> threads {};
|
||||
std::atomic<uint64_t> slots[129] {};
|
||||
threads.reserve(128);
|
||||
|
||||
auto acquire = [&](int idx) {
|
||||
++Waiters;
|
||||
|
||||
// Spin until allowed to race.
|
||||
while (!Acquire.load()) {
|
||||
// Be nice to valgrind.
|
||||
std::this_thread::yield();
|
||||
}
|
||||
|
||||
auto slot = set.allocate(1);
|
||||
|
||||
if (slot == set.invalid()) {
|
||||
// Set to invalid slot. Should never occur.
|
||||
slot = 128;
|
||||
}
|
||||
|
||||
// Increment the slot counter for the number of times this slot has allocated.
|
||||
slots[slot]++;
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < 128; ++i) {
|
||||
threads.emplace_back(acquire, i);
|
||||
}
|
||||
|
||||
// Wait until all threads are claimed to be ready.
|
||||
while (Waiters.load() != 128) {
|
||||
// Be nice to valgrind.
|
||||
std::this_thread::yield();
|
||||
}
|
||||
|
||||
Acquire = true;
|
||||
|
||||
// Wait for threads to exit.
|
||||
for (auto& t : threads) {
|
||||
t.join();
|
||||
}
|
||||
|
||||
// Every slot should only ever be acquired once.
|
||||
for (size_t i = 0; i < 128; ++i) {
|
||||
CHECK(slots[i].load() == 1);
|
||||
}
|
||||
|
||||
// There should be no invalid slots returned.
|
||||
CHECK(slots[128].load() == 0);
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
|
||||
TEST_CASE("Larger than word - rewind") {
|
||||
constexpr size_t size = 4096;
|
||||
auto buf = AllocateProtectedBuffer(size);
|
||||
|
||||
FEXCore::Utils::atomic_bitset<false, false> set {};
|
||||
|
||||
// Move the set up to the edge of the page to detect overruns.
|
||||
set.init(reinterpret_cast<uint8_t*>(buf.ptr) + (4096 - 32), 256);
|
||||
|
||||
REQUIRE(set.size_in_bits() == 256);
|
||||
|
||||
// Allocate all 256-bits
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
REQUIRE(set.allocate(1) != set.invalid());
|
||||
}
|
||||
|
||||
// Free the first 128-bits
|
||||
for (size_t i = 0; i < 128; ++i) {
|
||||
set.free(i, 1);
|
||||
}
|
||||
|
||||
std::atomic<bool> Running {};
|
||||
std::atomic<bool> Stop {};
|
||||
std::thread t {[&]() {
|
||||
LogMan::Msg::DFmt("Spinning");
|
||||
// Acquire and free 1-bit back to back
|
||||
while (!Stop) {
|
||||
auto slot = set.allocate(1);
|
||||
|
||||
// Introduce some variability by yielding here.
|
||||
std::this_thread::yield();
|
||||
|
||||
Running = true;
|
||||
if (slot != set.invalid()) {
|
||||
set.free(slot, 1);
|
||||
} else {
|
||||
LogMan::Msg::DFmt("We're full!");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}};
|
||||
|
||||
LogMan::Msg::DFmt("Waiting for thread to start!");
|
||||
while (!Running.load()) {
|
||||
// Be nice to valgrind.
|
||||
std::this_thread::yield();
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Attempting to allocate 128-bit while contended");
|
||||
|
||||
// Try and acquire 128-bits while contended.
|
||||
size_t attempts {};
|
||||
for (;;) {
|
||||
auto slot = set.allocate(128);
|
||||
if (slot == set.invalid()) {
|
||||
++attempts;
|
||||
// Be nice to valgrind.
|
||||
std::this_thread::yield();
|
||||
continue;
|
||||
}
|
||||
|
||||
// Can't fit in anything other than slot0
|
||||
REQUIRE(slot == 0);
|
||||
LogMan::Msg::DFmt("We got 128-bits in slot: {} after {} attempts", slot, attempts);
|
||||
set.free(slot, 128);
|
||||
break;
|
||||
}
|
||||
Stop = true;
|
||||
t.join();
|
||||
|
||||
FreeProtectedBuffer(buf);
|
||||
}
|
||||
@@ -1318,11 +1318,11 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - extended register") {
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 3), "cmn w28, w27, lsl #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 4), "cmn w28, w27, lsl #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 0), "cmn w28, x27");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 1), "cmn w28, x27, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 2), "cmn w28, x27, lsl #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 3), "cmn w28, x27, lsl #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 4), "cmn w28, x27, lsl #4");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 0), "cmn w28, x27, uxtx");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 1), "cmn w28, x27, uxtx #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 2), "cmn w28, x27, uxtx #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 3), "cmn w28, x27, uxtx #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 4), "cmn w28, x27, uxtx #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 0), "cmn w28, w27, sxtb");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 1), "cmn w28, w27, sxtb #1");
|
||||
@@ -1360,13 +1360,13 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - extended register") {
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 3), "cmn x28, w27, uxth #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 4), "cmn x28, w27, uxth #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 0), "cmn x28, w27, uxtw");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 0), "cmn x28, w27");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 1), "cmn x28, w27, uxtw #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 2), "cmn x28, w27, uxtw #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 3), "cmn x28, w27, uxtw #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 4), "cmn x28, w27, uxtw #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 0), "cmn x28, x27");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 0), "cmn x28, x27, lsl");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 1), "cmn x28, x27, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 2), "cmn x28, x27, lsl #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 3), "cmn x28, x27, lsl #3");
|
||||
@@ -1606,11 +1606,11 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - extended register") {
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_32, 3), "cmp w29, w28, lsl #3");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_32, 4), "cmp w29, w28, lsl #4");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 0), "cmp w29, x28");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 1), "cmp w29, x28, lsl #1");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 2), "cmp w29, x28, lsl #2");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 3), "cmp w29, x28, lsl #3");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 4), "cmp w29, x28, lsl #4");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 0), "cmp w29, x28, uxtx");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 1), "cmp w29, x28, uxtx #1");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 2), "cmp w29, x28, uxtx #2");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 3), "cmp w29, x28, uxtx #3");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 4), "cmp w29, x28, uxtx #4");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::SXTB, 0), "cmp w29, w28, sxtb");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r29, Reg::r28, ExtendedType::SXTB, 1), "cmp w29, w28, sxtb #1");
|
||||
@@ -1648,13 +1648,13 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - extended register") {
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTH, 3), "cmp x29, w28, uxth #3");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTH, 4), "cmp x29, w28, uxth #4");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTW, 0), "cmp x29, w28, uxtw");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTW, 0), "cmp x29, w28");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTW, 1), "cmp x29, w28, uxtw #1");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTW, 2), "cmp x29, w28, uxtw #2");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTW, 3), "cmp x29, w28, uxtw #3");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::UXTW, 4), "cmp x29, w28, uxtw #4");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 0), "cmp x29, x28");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 0), "cmp x29, x28, lsl");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 1), "cmp x29, x28, lsl #1");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 2), "cmp x29, x28, lsl #2");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r29, Reg::r28, ExtendedType::LSL_64, 3), "cmp x29, x28, lsl #3");
|
||||
|
||||
@@ -142,7 +142,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld2<SubRegSize::i32Bit>(DReg::d26, DReg::d27, Reg::r30), "ld2 {v26.2s, v27.2s}, [x30]");
|
||||
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(QReg::q26, QReg::q27, Reg::r30), "ld2 {v26.2d, v27.2d}, [x30]");
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30), "unallocated (NEONLoadStoreMultiStruct)");
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st2<SubRegSize::i8Bit>(QReg::q31, QReg::q0, Reg::r30), "st2 {v31.16b, v0.16b}, [x30]");
|
||||
TEST_SINGLE(st2<SubRegSize::i8Bit>(DReg::d31, DReg::d0, Reg::r30), "st2 {v31.8b, v0.8b}, [x30]");
|
||||
@@ -156,7 +156,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st2<SubRegSize::i32Bit>(DReg::d26, DReg::d27, Reg::r30), "st2 {v26.2s, v27.2s}, [x30]");
|
||||
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(QReg::q26, QReg::q27, Reg::r30), "st2 {v26.2d, v27.2d}, [x30]");
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30), "unallocated (NEONLoadStoreMultiStruct)");
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, Reg::r30), "ld3 {v31.16b, v0.16b, v1.16b}, [x30]");
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30), "ld3 {v31.8b, v0.8b, v1.8b}, [x30]");
|
||||
@@ -170,7 +170,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld3<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30), "ld3 {v26.2s, v27.2s, v28.2s}, [x30]");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, Reg::r30), "ld3 {v26.2d, v27.2d, v28.2d}, [x30]");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30), "unallocated (NEONLoadStoreMultiStruct)");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st3<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, Reg::r30), "st3 {v31.16b, v0.16b, v1.16b}, [x30]");
|
||||
TEST_SINGLE(st3<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30), "st3 {v31.8b, v0.8b, v1.8b}, [x30]");
|
||||
@@ -184,7 +184,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st3<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30), "st3 {v26.2s, v27.2s, v28.2s}, [x30]");
|
||||
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, Reg::r30), "st3 {v26.2d, v27.2d, v28.2d}, [x30]");
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30), "unallocated (NEONLoadStoreMultiStruct)");
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld4<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, QReg::q2, Reg::r30), "ld4 {v31.16b, v0.16b, v1.16b, v2.16b}, [x30]");
|
||||
TEST_SINGLE(ld4<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, DReg::d2, Reg::r30), "ld4 {v31.8b, v0.8b, v1.8b, v2.8b}, [x30]");
|
||||
@@ -199,7 +199,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld4<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30), "ld4 {v26.2s, v27.2s, v28.2s, v29.2s}, [x30]");
|
||||
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, QReg::q29, Reg::r30), "ld4 {v26.2d, v27.2d, v28.2d, v29.2d}, [x30]");
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30), "unallocated (NEONLoadStoreMultiStruct)");
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st4<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, QReg::q2, Reg::r30), "st4 {v31.16b, v0.16b, v1.16b, v2.16b}, [x30]");
|
||||
TEST_SINGLE(st4<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, DReg::d2, Reg::r30), "st4 {v31.8b, v0.8b, v1.8b, v2.8b}, [x30]");
|
||||
@@ -214,7 +214,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st4<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30), "st4 {v26.2s, v27.2s, v28.2s, v29.2s}, [x30]");
|
||||
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, QReg::q29, Reg::r30), "st4 {v26.2d, v27.2d, v28.2d, v29.2d}, [x30]");
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30), "unallocated (NEONLoadStoreMultiStruct)");
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30), "unallocated (Unallocated)");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store multiple structures (post-indexed)") {
|
||||
@@ -486,7 +486,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld2<SubRegSize::i32Bit>(DReg::d26, DReg::d27, Reg::r30, Reg::r29), "ld2 {v26.2s, v27.2s}, [x30], x29");
|
||||
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(QReg::q26, QReg::q27, Reg::r30, Reg::r29), "ld2 {v26.2d, v27.2d}, [x30], x29");
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, Reg::r29), "unallocated (NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, Reg::r29), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld2<SubRegSize::i8Bit>(QReg::q31, QReg::q0, Reg::r30, 32), "ld2 {v31.16b, v0.16b}, [x30], #32");
|
||||
TEST_SINGLE(ld2<SubRegSize::i8Bit>(DReg::d31, DReg::d0, Reg::r30, 16), "ld2 {v31.8b, v0.8b}, [x30], #16");
|
||||
@@ -500,7 +500,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld2<SubRegSize::i32Bit>(DReg::d26, DReg::d27, Reg::r30, 16), "ld2 {v26.2s, v27.2s}, [x30], #16");
|
||||
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(QReg::q26, QReg::q27, Reg::r30, 32), "ld2 {v26.2d, v27.2d}, [x30], #32");
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, 16), "unallocated (NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(ld2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, 16), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st2<SubRegSize::i8Bit>(QReg::q31, QReg::q0, Reg::r30, Reg::r29), "st2 {v31.16b, v0.16b}, [x30], x29");
|
||||
TEST_SINGLE(st2<SubRegSize::i8Bit>(DReg::d31, DReg::d0, Reg::r30, Reg::r29), "st2 {v31.8b, v0.8b}, [x30], x29");
|
||||
@@ -514,7 +514,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st2<SubRegSize::i32Bit>(DReg::d26, DReg::d27, Reg::r30, Reg::r29), "st2 {v26.2s, v27.2s}, [x30], x29");
|
||||
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(QReg::q26, QReg::q27, Reg::r30, Reg::r29), "st2 {v26.2d, v27.2d}, [x30], x29");
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, Reg::r29), "unallocated (NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, Reg::r29), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st2<SubRegSize::i8Bit>(QReg::q31, QReg::q0, Reg::r30, 32), "st2 {v31.16b, v0.16b}, [x30], #32");
|
||||
TEST_SINGLE(st2<SubRegSize::i8Bit>(DReg::d31, DReg::d0, Reg::r30, 16), "st2 {v31.8b, v0.8b}, [x30], #16");
|
||||
@@ -528,7 +528,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st2<SubRegSize::i32Bit>(DReg::d26, DReg::d27, Reg::r30, 16), "st2 {v26.2s, v27.2s}, [x30], #16");
|
||||
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(QReg::q26, QReg::q27, Reg::r30, 32), "st2 {v26.2d, v27.2d}, [x30], #32");
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, 16), "unallocated (NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(st2<SubRegSize::i64Bit>(DReg::d26, DReg::d27, Reg::r30, 16), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, Reg::r30, Reg::r29), "ld3 {v31.16b, v0.16b, v1.16b}, [x30], x29");
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30, Reg::r29), "ld3 {v31.8b, v0.8b, v1.8b}, [x30], x29");
|
||||
@@ -542,8 +542,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld3<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, Reg::r29), "ld3 {v26.2s, v27.2s, v28.2s}, [x30], x29");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, Reg::r30, Reg::r29), "ld3 {v26.2d, v27.2d, v28.2d}, [x30], x29");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, Reg::r29), "unallocated "
|
||||
"(NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, Reg::r29), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, Reg::r30, 48), "ld3 {v31.16b, v0.16b, v1.16b}, [x30], #48");
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30, 24), "ld3 {v31.8b, v0.8b, v1.8b}, [x30], #24");
|
||||
@@ -557,7 +556,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld3<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 24), "ld3 {v26.2s, v27.2s, v28.2s}, [x30], #24");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, Reg::r30, 48), "ld3 {v26.2d, v27.2d, v28.2d}, [x30], #48");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 24), "unallocated (NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 24), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st3<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, Reg::r30, Reg::r29), "st3 {v31.16b, v0.16b, v1.16b}, [x30], x29");
|
||||
TEST_SINGLE(st3<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30, Reg::r29), "st3 {v31.8b, v0.8b, v1.8b}, [x30], x29");
|
||||
@@ -571,8 +570,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st3<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, Reg::r29), "st3 {v26.2s, v27.2s, v28.2s}, [x30], x29");
|
||||
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, Reg::r30, Reg::r29), "st3 {v26.2d, v27.2d, v28.2d}, [x30], x29");
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, Reg::r29), "unallocated "
|
||||
"(NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, Reg::r29), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st3<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, Reg::r30, 48), "st3 {v31.16b, v0.16b, v1.16b}, [x30], #48");
|
||||
TEST_SINGLE(st3<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30, 24), "st3 {v31.8b, v0.8b, v1.8b}, [x30], #24");
|
||||
@@ -586,7 +584,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(st3<SubRegSize::i32Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 24), "st3 {v26.2s, v27.2s, v28.2s}, [x30], #24");
|
||||
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, Reg::r30, 48), "st3 {v26.2d, v27.2d, v28.2d}, [x30], #48");
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 24), "unallocated (NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(st3<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 24), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld4<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, QReg::q2, Reg::r30, Reg::r29), "ld4 {v31.16b, v0.16b, v1.16b, v2.16b}, "
|
||||
"[x30], x29");
|
||||
@@ -609,9 +607,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, QReg::q29, Reg::r30, Reg::r29), "ld4 {v26.2d, v27.2d, v28.2d, "
|
||||
"v29.2d}, [x30], x29");
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, Reg::r29), "unallocated "
|
||||
"(NEONLoadStoreMultiStructPostIndex"
|
||||
")");
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, Reg::r29), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(ld4<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, QReg::q2, Reg::r30, 64), "ld4 {v31.16b, v0.16b, v1.16b, v2.16b}, "
|
||||
"[x30], #64");
|
||||
@@ -634,8 +630,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, QReg::q29, Reg::r30, 64), "ld4 {v26.2d, v27.2d, v28.2d, v29.2d}, "
|
||||
"[x30], #64");
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, 32), "unallocated "
|
||||
"(NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(ld4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, 32), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st4<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, QReg::q2, Reg::r30, Reg::r29), "st4 {v31.16b, v0.16b, v1.16b, v2.16b}, "
|
||||
"[x30], x29");
|
||||
@@ -658,9 +653,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, QReg::q29, Reg::r30, Reg::r29), "st4 {v26.2d, v27.2d, v28.2d, "
|
||||
"v29.2d}, [x30], x29");
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, Reg::r29), "unallocated "
|
||||
"(NEONLoadStoreMultiStructPostIndex"
|
||||
")");
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, Reg::r29), "unallocated (Unallocated)");
|
||||
|
||||
TEST_SINGLE(st4<SubRegSize::i8Bit>(QReg::q31, QReg::q0, QReg::q1, QReg::q2, Reg::r30, 64), "st4 {v31.16b, v0.16b, v1.16b, v2.16b}, "
|
||||
"[x30], #64");
|
||||
@@ -683,8 +676,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(QReg::q26, QReg::q27, QReg::q28, QReg::q29, Reg::r30, 64), "st4 {v26.2d, v27.2d, v28.2d, v29.2d}, "
|
||||
"[x30], #64");
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, 32), "unallocated "
|
||||
"(NEONLoadStoreMultiStructPostIndex)");
|
||||
TEST_SINGLE(st4<SubRegSize::i64Bit>(DReg::d26, DReg::d27, DReg::d28, DReg::d29, Reg::r30, 32), "unallocated (Unallocated)");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: ASIMD loadstore single") {
|
||||
TEST_SINGLE(ld1<SubRegSize::i8Bit>(VReg::v26, 0, Reg::r30), "ld1 {v26.b}[0], [x30]");
|
||||
@@ -2284,15 +2276,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(staddl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "staddl w30, [x29]");
|
||||
TEST_SINGLE(staddl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "staddl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stadda(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddab w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i16Bit, Reg::r30, Reg::r29), "staddah w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stadda w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stadda x30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldaddab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldaddah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldadda w30, wzr, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldadda x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(staddal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddalb w30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "staddalh w30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "staddal w30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "staddal x30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldaddalb w30, wzr, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldaddalh w30, wzr, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldaddal w30, wzr, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldaddal x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stclr(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrb w30, [x29]");
|
||||
TEST_SINGLE(stclr(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclrh w30, [x29]");
|
||||
@@ -2304,15 +2296,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(stclrl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclrl w30, [x29]");
|
||||
TEST_SINGLE(stclrl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclrl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stclra(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrab w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclrah w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclra w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclra x30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldclrab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldclrah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldclra w30, wzr, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldclra x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stclral(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclralb w30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclralh w30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclral w30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclral x30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldclralb w30, wzr, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldclralh w30, wzr, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldclral w30, wzr, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldclral x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stset(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetb w30, [x29]");
|
||||
TEST_SINGLE(stset(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stseth w30, [x29]");
|
||||
@@ -2324,15 +2316,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(stsetl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsetl w30, [x29]");
|
||||
TEST_SINGLE(stsetl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsetl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stseta(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetab w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsetah w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stseta w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stseta x30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldsetab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldsetah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldseta w30, wzr, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldseta x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stsetal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetalb w30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsetalh w30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsetal w30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsetal x30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldsetalb w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldsetalh w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldsetal w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldsetal x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(steor(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorb w30, [x29]");
|
||||
TEST_SINGLE(steor(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steorh w30, [x29]");
|
||||
@@ -2344,15 +2336,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(steorl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steorl w30, [x29]");
|
||||
TEST_SINGLE(steorl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steorl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(steora(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorab w30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steorah w30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steora w30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steora x30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldeorab w30, wzr, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldeorah w30, wzr, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldeora w30, wzr, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldeora x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(steoral(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steoralb w30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steoralh w30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steoral w30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steoral x30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldeoralb w30, wzr, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldeoralh w30, wzr, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldeoral w30, wzr, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldeoral x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmax(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxb w30, [x29]");
|
||||
TEST_SINGLE(stsmax(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxh w30, [x29]");
|
||||
@@ -2364,15 +2356,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(stsmaxl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmaxl w30, [x29]");
|
||||
TEST_SINGLE(stsmaxl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmaxl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxab w30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxah w30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmaxa w30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmaxa x30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldsmaxab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldsmaxah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldsmaxa w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldsmaxa x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxalb w30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxalh w30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmaxal w30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmaxal x30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldsmaxalb w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldsmaxalh w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldsmaxal w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldsmaxal x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmin(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminb w30, [x29]");
|
||||
TEST_SINGLE(stsmin(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminh w30, [x29]");
|
||||
@@ -2384,15 +2376,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(stsminl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsminl w30, [x29]");
|
||||
TEST_SINGLE(stsminl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsminl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmina(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminab w30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminah w30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmina w30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmina x30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldsminab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldsminah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldsmina w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldsmina x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stsminal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminalb w30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminalh w30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsminal w30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsminal x30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldsminalb w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldsminalh w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldsminal w30, wzr, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldsminal x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stumax(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxb w30, [x29]");
|
||||
TEST_SINGLE(stumax(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxh w30, [x29]");
|
||||
@@ -2404,15 +2396,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(stumaxl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumaxl w30, [x29]");
|
||||
TEST_SINGLE(stumaxl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumaxl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxab w30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxah w30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumaxa w30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumaxa x30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldumaxab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldumaxah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldumaxa w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldumaxa x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxalb w30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxalh w30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumaxal w30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumaxal x30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "ldumaxalb w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "ldumaxalh w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldumaxal w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldumaxal x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stumin(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminb w30, [x29]");
|
||||
TEST_SINGLE(stumin(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminh w30, [x29]");
|
||||
@@ -2424,15 +2416,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations
|
||||
TEST_SINGLE(stuminl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stuminl w30, [x29]");
|
||||
TEST_SINGLE(stuminl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stuminl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumina(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminab w30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminah w30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumina w30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumina x30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i8Bit, Reg::r30, Reg::r29), "lduminab w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i16Bit, Reg::r30, Reg::r29), "lduminah w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i32Bit, Reg::r30, Reg::r29), "ldumina w30, wzr, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i64Bit, Reg::r30, Reg::r29), "ldumina x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(stuminal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminalb w30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminalh w30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stuminal w30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stuminal x30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "lduminalb w30, wzr, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "lduminalh w30, wzr, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "lduminal w30, wzr, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "lduminal x30, xzr, [x29]");
|
||||
|
||||
TEST_SINGLE(ldswp(SubRegSize::i8Bit, Reg::r30, Reg::r28, Reg::r29), "swpb w30, w28, [x29]");
|
||||
TEST_SINGLE(ldswp(SubRegSize::i16Bit, Reg::r30, Reg::r28, Reg::r29), "swph w30, w28, [x29]");
|
||||
|
||||
@@ -294,7 +294,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point convert pre
|
||||
/////< Size is destination size
|
||||
// void fcvtlt(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, ARMEmitter::PRegister pg, ARMEmitter::ZRegister zn) {
|
||||
|
||||
// XXX: BFCVTNT
|
||||
TEST_SINGLE(bfcvtnt(ZReg::z30, PReg::p6.Merging(), ZReg::z29), "bfcvtnt z30.h, p6/m, z29.s");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE2 floating-point pairwise operations") {
|
||||
// TEST_SINGLE(faddp(SubRegSize::i8Bit, ZReg::z30, PReg::p6.Merging(), ZReg::z30, ZReg::z28), "faddp z30.b, p6/m, z30.b, z28.b");
|
||||
@@ -3420,17 +3420,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point multiply-ad
|
||||
TEST_SINGLE(fmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 3), "fmlslt z30.s, z29.h, z7.h[3]");
|
||||
TEST_SINGLE(fmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 7), "fmlslt z30.s, z29.h, z7.h[7]");
|
||||
|
||||
TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 0), "bfmlalb z30.s, z29.h, z7.h[0]");
|
||||
TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 3), "bfmlalb z30.s, z29.h, z7.h[3]");
|
||||
TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 7), "bfmlalb z30.s, z29.h, z7.h[7]");
|
||||
|
||||
TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 0), "bfmlalt z30.s, z29.h, z7.h[0]");
|
||||
TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 3), "bfmlalt z30.s, z29.h, z7.h[3]");
|
||||
TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 7), "bfmlalt z30.s, z29.h, z7.h[7]");
|
||||
|
||||
// XXX: vixl's diassembler doesn't support these. Re-enable when it does
|
||||
// or upon switching disassemblers.
|
||||
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 0), "bfmlalb z30.s, z29.h, z7.h[0]");
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 3), "bfmlalb z30.s, z29.h, z7.h[3]");
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 7), "bfmlalb z30.s, z29.h, z7.h[7]");
|
||||
|
||||
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 0), "bfmlalt z30.s, z29.h, z7.h[0]");
|
||||
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 3), "bfmlalt z30.s, z29.h, z7.h[3]");
|
||||
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 7), "bfmlalt z30.s, z29.h, z7.h[7]");
|
||||
|
||||
// TEST_SINGLE(bfmlslb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 0), "bfmlslb z30.s, z29.h, z7.h[0]");
|
||||
// TEST_SINGLE(bfmlslb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 3), "bfmlslb z30.s, z29.h, z7.h[3]");
|
||||
// TEST_SINGLE(bfmlslb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z7, 7), "bfmlslb z30.s, z29.h, z7.h[7]");
|
||||
@@ -3459,20 +3459,20 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point multiply-ad
|
||||
TEST_SINGLE(fmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "fmlslt z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(fmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "fmlslt z30.s, z29.h, z28.h");
|
||||
|
||||
TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
|
||||
|
||||
TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
|
||||
|
||||
// XXX: vixl's diassembler doesn't support these. Re-enable when it does
|
||||
// or upon switching disassemblers.
|
||||
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
|
||||
|
||||
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
|
||||
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlslb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlslb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlslb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
|
||||
|
||||
// TEST_SINGLE(bfmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslt z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(bfmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslt z30.s, z29.h, z28.h");
|
||||
@@ -3937,9 +3937,12 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE load and broadcast element
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i8Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.b}, p6/z, [x29, #31]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i8Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.b}, p6/z, [x29, #63]");
|
||||
|
||||
// TODO: Several instances are commented out due to a reported bug in the vixl dissassembler.
|
||||
// Uncomment these when it's fixed.
|
||||
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rb {z30.h}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.h}, p6/z, [x29, #31]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.h}, p6/z, [x29, #63]");
|
||||
// TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.h}, p6/z, [x29, #31]");
|
||||
// TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.h}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rb {z30.s}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.s}, p6/z, [x29, #31]");
|
||||
@@ -3950,8 +3953,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE load and broadcast element
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.d}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rsb {z30.h}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rsb {z30.h}, p6/z, [x29, #31]");
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rsb {z30.h}, p6/z, [x29, #63]");
|
||||
// TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rsb {z30.h}, p6/z, [x29, #31]");
|
||||
// TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rsb {z30.h}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rsb {z30.s}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rsb {z30.s}, p6/z, [x29, #31]");
|
||||
@@ -3962,8 +3965,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE load and broadcast element
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rsb {z30.d}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rh {z30.h}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 64), "ld1rh {z30.h}, p6/z, [x29, #64]");
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 126), "ld1rh {z30.h}, p6/z, [x29, #126]");
|
||||
// TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 64), "ld1rh {z30.h}, p6/z, [x29, #64]");
|
||||
// TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 126), "ld1rh {z30.h}, p6/z, [x29, #126]");
|
||||
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rh {z30.s}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 64), "ld1rh {z30.s}, p6/z, [x29, #64]");
|
||||
@@ -4389,6 +4392,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point convert pre
|
||||
TEST_SINGLE(fcvt(SubRegSize::i64Bit, SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), ZReg::z29), "fcvt z30.d, p6/m, z29.s");
|
||||
|
||||
TEST_SINGLE(fcvtx(ZReg::z30, PReg::p6.Merging(), ZReg::z29), "fcvtx z30.s, p6/m, z29.d");
|
||||
|
||||
TEST_SINGLE(bfcvt(ZReg::z30, PReg::p6.Merging(), ZReg::z29), "bfcvt z30.h, p6/m, z29.s");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point unary operations") {
|
||||
|
||||
@@ -660,22 +660,25 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Advanced SIMD scalar x inde
|
||||
TEST_SINGLE(sqrdmulh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "sqrdmulh s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(sqrdmulh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "sqrdmulh s30, s29, v28.s[3]");
|
||||
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmla h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmla h30, h29, v15.h[7]");
|
||||
// TODO: Commented out due to a bug in vixl's decoder (which has been reported).
|
||||
// Uncomment these when fixed.
|
||||
|
||||
// TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmla h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmla h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmla s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmla s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmla d30, d29, v28.d[0]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 1), "fmla d30, d29, v28.d[1]");
|
||||
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmls h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmls h30, h29, v15.h[7]");
|
||||
// TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmls h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmls h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmls s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmls s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmls d30, d29, v28.d[0]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 1), "fmls d30, d29, v28.d[1]");
|
||||
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmul h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmul h30, h29, v15.h[7]");
|
||||
// TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmul h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmul h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmul s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmul s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmul d30, d29, v28.d[0]");
|
||||
@@ -691,8 +694,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Advanced SIMD scalar x inde
|
||||
TEST_SINGLE(sqrdmlsh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "sqrdmlsh s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(sqrdmlsh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "sqrdmlsh s30, s29, v28.s[3]");
|
||||
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmulx h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmulx h30, h29, v15.h[7]");
|
||||
// TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmulx h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmulx h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmulx s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmulx s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmulx d30, d29, v28.d[0]");
|
||||
|
||||
@@ -67,11 +67,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: Exception generation") {
|
||||
TEST_SINGLE(dcps3(65535), "dcps3 {#0xffff}");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System instructions with register argument") {
|
||||
if (false) {
|
||||
// Unsupported in vixl.
|
||||
TEST_SINGLE(wfet(Reg::r30), "wfet x30");
|
||||
TEST_SINGLE(wfit(Reg::r30), "wfit x30");
|
||||
}
|
||||
TEST_SINGLE(wfet(Reg::r30), "wfet x30");
|
||||
TEST_SINGLE(wfit(Reg::r30), "wfit x30");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: System: Hints") {
|
||||
TEST_SINGLE(nop(), "nop");
|
||||
@@ -116,7 +113,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: Barriers") {
|
||||
TEST_SINGLE(isb(), "isb");
|
||||
|
||||
TEST_SINGLE(sb(), "sb");
|
||||
TEST_SINGLE(tcommit(), "tcommit");
|
||||
TEST_SINGLE(tcommit(), "unimplemented (Unimplemented)");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System register move") {
|
||||
// vixl doesn't have decoding for a bunch of these.
|
||||
|
||||
@@ -96,6 +96,7 @@ def GetHostFeatures(data):
|
||||
return HostFeaturesData
|
||||
|
||||
def parse_json_data(json_filepath, json_filename, json_data, output_binary_path):
|
||||
BinaryCacheVersion = 0
|
||||
Bitness = 64
|
||||
EnabledHostFeatures = HostFeatures.FEATURE_ANY
|
||||
DisabledHostFeatures = HostFeatures.FEATURE_ANY
|
||||
@@ -106,6 +107,9 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
if ("Bitness" in items):
|
||||
Bitness = int(items["Bitness"])
|
||||
|
||||
if ("BinaryCacheVersion" in items):
|
||||
BinaryCacheVersion = int(items["BinaryCacheVersion"])
|
||||
|
||||
if ("EnabledHostFeatures" in items):
|
||||
EnabledHostFeatures = GetHostFeatures(items["EnabledHostFeatures"])
|
||||
|
||||
@@ -177,6 +181,7 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
# struct TestInfo;
|
||||
# struct DataHeader {
|
||||
# uint64_t Bitness;
|
||||
# uint64_t BinaryCacheVersion;
|
||||
# uint64_t NumTests;
|
||||
# uint64_t EnabledHostFeatures;
|
||||
# uint64_t DisabledHostFeatures;
|
||||
@@ -197,6 +202,7 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
|
||||
# Add the header
|
||||
MemData += struct.pack('Q', Bitness)
|
||||
MemData += struct.pack('Q', BinaryCacheVersion)
|
||||
MemData += struct.pack('Q', len(TestDataMap))
|
||||
MemData += struct.pack('Q', EnabledHostFeatures.value)
|
||||
MemData += struct.pack('Q', DisabledHostFeatures.value)
|
||||
|
||||
@@ -10,7 +10,10 @@ def insert_before(d, key, item):
|
||||
items.insert(list(d.keys()).index(key), item)
|
||||
return dict(items)
|
||||
|
||||
def update_performance_numbers(performance_json_path, performance_json, new_json_numbers):
|
||||
def update_performance_numbers(performance_json_path, performance_json, new_json_numbers, new_cache_version):
|
||||
|
||||
performance_json["Features"]["BinaryCacheVersion"] = new_cache_version
|
||||
|
||||
for key, items in new_json_numbers.items():
|
||||
if len(key) == 0:
|
||||
continue
|
||||
@@ -64,11 +67,19 @@ def main():
|
||||
if not isinstance(performance_json_data, dict):
|
||||
raise TypeError('JSON data must be a dict')
|
||||
|
||||
new_json_numbers_data = json.loads(new_json_numbers_text)
|
||||
new_json_numbers_base = json.loads(new_json_numbers_text)
|
||||
|
||||
features = new_json_numbers_base["Features"]
|
||||
if not isinstance(features, dict):
|
||||
raise TypeError('features JSON data must be a dict')
|
||||
|
||||
new_cache_version = features["BinaryCacheVersion"]
|
||||
|
||||
new_json_numbers_data = new_json_numbers_base["Instructions"]
|
||||
if not isinstance(new_json_numbers_data, dict):
|
||||
raise TypeError('JSON data must be a dict')
|
||||
|
||||
return update_performance_numbers(performance_json_path, performance_json_data, new_json_numbers_data)
|
||||
return update_performance_numbers(performance_json_path, performance_json_data, new_json_numbers_data, new_cache_version)
|
||||
except ValueError as ve:
|
||||
logging.error(f'JSON error: {ve}')
|
||||
return 1
|
||||
|
||||
@@ -75,6 +75,14 @@ BigCoreIDs = {
|
||||
tuple([0x51, 0x800]): "cortex-a73", # Kryo 2xx Gold
|
||||
tuple([0x51, 0x802]): "cortex-a75", # Kryo 3xx Gold
|
||||
tuple([0x51, 0x804]): "cortex-a76", # Kryo 4xx Gold
|
||||
tuple([0x51, 0x001]): # Oryon-1
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["oryon-1", "19.0"], # Only exists in clang 19.0 or newer.
|
||||
],
|
||||
tuple([0x51, 0x002]): # Oryon-3 is unknown to clang for now.
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["oryon-1", "19.0"], # Only exists in clang 19.0 or newer.
|
||||
],
|
||||
# Apple M1 Parallels hypervisor
|
||||
tuple([0x41, 0x0]):
|
||||
[ ["apple-a13", "0.0"], # If we aren't on 12.0+
|
||||
|
||||
@@ -20,6 +20,8 @@
|
||||
import re
|
||||
from pathlib import Path
|
||||
import sys
|
||||
import traceback
|
||||
import logging
|
||||
|
||||
if (len(sys.argv) != 4):
|
||||
print("doc_outline_generator GIT_DIR SRC_DIR LINK_PREFIX")
|
||||
@@ -62,17 +64,29 @@ for path in Paths:
|
||||
name = entry.split(":", 1)[0].strip();
|
||||
val = entry.split(":", 1)[1].strip();
|
||||
if name == "category":
|
||||
cat_name = val.split("~", 1)[0].strip();
|
||||
cat_val = val.split("~", 1)[1].strip();
|
||||
CategoryLabels[cat_name] = cat_val
|
||||
try:
|
||||
cat_name = val.split("~", 1)[0].strip();
|
||||
cat_val = val.split("~", 1)[1].strip();
|
||||
CategoryLabels[cat_name] = cat_val
|
||||
except Exception as e:
|
||||
logging.error("Failure to parse {}".format(val))
|
||||
logging.error(traceback.format_exc())
|
||||
elif name == "meta":
|
||||
meta_name = val.split("~", 1)[0].strip();
|
||||
meta_val = val.split("~", 1)[1].strip();
|
||||
MetaLabels[meta_name] = meta_val
|
||||
try:
|
||||
meta_name = val.split("~", 1)[0].strip();
|
||||
meta_val = val.split("~", 1)[1].strip();
|
||||
MetaLabels[meta_name] = meta_val
|
||||
except Exception as e:
|
||||
logging.error("Failure to parse {}".format(val))
|
||||
logging.error(traceback.format_exc())
|
||||
elif name == "glossary":
|
||||
glossary_name = val.split("~", 1)[0].strip();
|
||||
glossary_val = val.split("~", 1)[1].strip();
|
||||
GlossaryLabels[glossary_name] = glossary_val
|
||||
try:
|
||||
glossary_name = val.split("~", 1)[0].strip();
|
||||
glossary_val = val.split("~", 1)[1].strip();
|
||||
GlossaryLabels[glossary_name] = glossary_val
|
||||
except Exception as e:
|
||||
logging.error("Failure to parse {}".format(val))
|
||||
logging.error(traceback.format_exc())
|
||||
elif name == "tags":
|
||||
for meta_name in val.split(","):
|
||||
if meta_name.strip() not in Meta:
|
||||
|
||||
+2
-2
@@ -35,13 +35,13 @@ if [ "$CHANGED_ONLY" = true ]; then
|
||||
|
||||
CHANGED_FILES=$(git ls-files -m '*.cpp' '*.h' '*.inl')
|
||||
if [ -n "$CHANGED_FILES" ]; then
|
||||
echo "$CHANGED_FILES" | xargs -d '\n' -n 1 -P "$(nproc)" clang-format-19 -i
|
||||
echo "$CHANGED_FILES" | xargs -d '\n' -n 1 -P "$(nproc)" $CLANG_FORMAT -i
|
||||
else
|
||||
echo "No changed files to format."
|
||||
fi
|
||||
else
|
||||
# Reformat whole tree (original behavior)
|
||||
git ls-files -z '*.cpp' '*.h' '*.inl' | xargs -0 -n 1 -P "$(nproc)" clang-format-19 -i
|
||||
git ls-files -z '*.cpp' '*.h' '*.inl' | xargs -0 -n 1 -P "$(nproc)" $CLANG_FORMAT -i
|
||||
fi
|
||||
|
||||
cd "$DIR"
|
||||
@@ -416,16 +416,17 @@ static fextl::string RecoverGuestProgramFilename(fextl::string Program, bool Exe
|
||||
// Only in the case that FEX is executing an FD will the program argument potentially be a symlink.
|
||||
// This symlink will be in the style of `/dev/fd/<FD>`.
|
||||
//
|
||||
// If the argument /is/ a symlink then resolve its path to get the original application name.
|
||||
if (FHU::Symlinks::IsSymlink(Program)) {
|
||||
// If the argument /is/ a symlink then resolve its path continuously until the original application name.
|
||||
while (FHU::Symlinks::IsSymlink(Program)) {
|
||||
char Filename[PATH_MAX];
|
||||
auto SymlinkPath = FHU::Symlinks::ResolveSymlink(Program, Filename);
|
||||
if (SymlinkPath.starts_with('/')) {
|
||||
// This file was executed through an FD.
|
||||
// Remove the ` (deleted)` text if the file was deleted after the fact.
|
||||
// Otherwise just get the symlink without the deleted text.
|
||||
return fextl::string {SymlinkPath.substr(0, SymlinkPath.rfind(" (deleted)"))};
|
||||
if (!SymlinkPath.starts_with('/')) {
|
||||
break;
|
||||
}
|
||||
// This file was executed through an FD.
|
||||
// Remove the ` (deleted)` text if the file was deleted after the fact.
|
||||
// Otherwise just get the symlink without the deleted text.
|
||||
Program = fextl::string {SymlinkPath.substr(0, SymlinkPath.rfind(" (deleted)"))};
|
||||
}
|
||||
}
|
||||
#endif
|
||||
@@ -713,7 +714,6 @@ fextl::string GetCacheDirectory() {
|
||||
return CacheOverride;
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
#ifdef FEX_STEAM_SUPPORT
|
||||
const char* SteamDataPath = getenv("STEAM_COMPAT_SHADER_PATH");
|
||||
if (SteamDataPath) {
|
||||
@@ -721,6 +721,7 @@ fextl::string GetCacheDirectory() {
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
auto HomeDir = GetHomeDirectory();
|
||||
const char* CacheXDG = getenv("XDG_CACHE_HOME");
|
||||
return (CacheXDG ? fextl::string {CacheXDG} : (fextl::string {HomeDir} + "/.cache")) + "/fex-emu/";
|
||||
@@ -741,5 +742,6 @@ void InitializeConfigs(const PortableInformation& PortableInfo) {
|
||||
FEXCore::Config::SetConfigDirectory(GetConfigDirectory(true, PortableInfo), true);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(false, PortableInfo), false);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(true, PortableInfo), true);
|
||||
FEXCore::Config::SetCacheDirectory(GetCacheDirectory());
|
||||
}
|
||||
} // namespace FEX::Config
|
||||
@@ -72,8 +72,10 @@ GetSysReg(MIDR_EL1, MIDR_EL1);
|
||||
GetSysReg(ISAR1_EL1, ID_AA64ISAR1_EL1);
|
||||
GetSysReg(MMFR0_EL1, ID_AA64MMFR0_EL1);
|
||||
GetSysReg(MMFR2_EL1, ID_AA64MMFR2_EL1);
|
||||
#ifndef _WIN32
|
||||
GetSysReg(MMFR3_EL1, s3_0_c0_c7_3); // Can't request by name
|
||||
GetSysReg(ZFR0_EL1, s3_0_c0_c4_4); // Can't request by name
|
||||
#endif
|
||||
GetSysReg(ZFR0_EL1, s3_0_c0_c4_4); // Can't request by name
|
||||
GetSysReg(MMFR1_EL1, ID_AA64MMFR1_EL1);
|
||||
GetSysReg(ISAR2_EL1, ID_AA64ISAR2_EL1);
|
||||
GetSysReg(DCZID_EL0, DCZID_EL0);
|
||||
@@ -541,11 +543,10 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
HostFeatures.PreferZVAForVZero = false;
|
||||
|
||||
if (CTR) {
|
||||
HostFeatures.DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
HostFeatures.ICacheLineSize = 4 << (CTR & 0xF);
|
||||
HostFeatures.DCacheLineLog2 = (CTR >> 16) & 0xF;
|
||||
} else {
|
||||
HostFeatures.DCacheLineSize = 64;
|
||||
HostFeatures.ICacheLineSize = 64;
|
||||
// 64-bytes
|
||||
HostFeatures.DCacheLineLog2 = 4;
|
||||
}
|
||||
|
||||
if (!HostFeatures.SupportsAtomics) {
|
||||
|
||||
+53
-24
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "FEXCore/Utils/StringUtils.h"
|
||||
#include "FEXHeaderUtils/Filesystem.h"
|
||||
|
||||
#include <optional>
|
||||
#include <stdlib.h>
|
||||
#include <tiny-json.h>
|
||||
|
||||
@@ -80,11 +81,11 @@ fextl::string GenerateSteamAppConfig(const FEX::Config::PortableInformation& Por
|
||||
|
||||
// Current supported Steam options.
|
||||
struct SteamOptions {
|
||||
bool TSO = true;
|
||||
bool Multiblock = true;
|
||||
bool Thunks_GL = false;
|
||||
bool Thunks_Vulkan = false;
|
||||
bool EnableLogging = false;
|
||||
std::optional<bool> TSO {};
|
||||
std::optional<bool> Multiblock {};
|
||||
std::optional<bool> Thunks_GL {};
|
||||
std::optional<bool> Thunks_Vulkan {};
|
||||
std::optional<bool> EnableLogging {};
|
||||
};
|
||||
SteamOptions Options {};
|
||||
|
||||
@@ -108,18 +109,35 @@ fextl::string GenerateSteamAppConfig(const FEX::Config::PortableInformation& Por
|
||||
const auto steam_fex_compat = getenv("STEAM_COMPAT_FEX_CONFIG");
|
||||
if (steam_fex_compat) {
|
||||
const auto steam_fex_compat_view = std::string_view(steam_fex_compat);
|
||||
if (steam_fex_compat_view.find("TSOEnabled:1") != steam_fex_compat_view.npos) {
|
||||
Options.TSO = true;
|
||||
}
|
||||
if (steam_fex_compat_view.find("Multiblock:1") != steam_fex_compat_view.npos) {
|
||||
Options.Multiblock = true;
|
||||
}
|
||||
if (steam_fex_compat_view.find("ThunksDB_GL:1") != steam_fex_compat_view.npos) {
|
||||
Options.Thunks_GL = true;
|
||||
}
|
||||
if (steam_fex_compat_view.find("ThunksDB_Vulkan:1") != steam_fex_compat_view.npos) {
|
||||
Options.Thunks_Vulkan = true;
|
||||
}
|
||||
|
||||
// Return a tristate for Steam options.
|
||||
// - Exists: Value must be 1 or 0 to set the boolean.
|
||||
// - Doesn't exist: std::optional is unset and config option isn't emitted.
|
||||
auto get_bool_flag = [steam_fex_compat_view](std::string_view option) -> std::optional<bool> {
|
||||
const auto pos = steam_fex_compat_view.find(option);
|
||||
const auto value_pos = pos + option.size();
|
||||
|
||||
// If the key isn't found, or the value position would be beyond the end of the view, then it's not set.
|
||||
if (pos == steam_fex_compat_view.npos || value_pos >= steam_fex_compat_view.size()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const auto value = steam_fex_compat_view[value_pos];
|
||||
if (value == '1') {
|
||||
return true;
|
||||
} else if (value == '0') {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Give an error on invalid boolean key.
|
||||
fextl::fmt::print(stderr, "Invalid option for bool config {}. Got {}\n", option, value);
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
Options.TSO = get_bool_flag("TSOEnabled:");
|
||||
Options.Multiblock = get_bool_flag("Multiblock:");
|
||||
Options.Thunks_GL = get_bool_flag("ThunksDB_GL:");
|
||||
Options.Thunks_Vulkan = get_bool_flag("ThunksDB_Vulkan:");
|
||||
}
|
||||
|
||||
// Create the json.
|
||||
@@ -128,19 +146,30 @@ fextl::string GenerateSteamAppConfig(const FEX::Config::PortableInformation& Por
|
||||
Dest = json_objOpen(Buffer, nullptr);
|
||||
{
|
||||
Dest = json_objOpen(Dest, "Config");
|
||||
Dest = json_str(Dest, "TSOEnabled", Options.TSO ? "1" : "0");
|
||||
Dest = json_str(Dest, "Multiblock", Options.Multiblock ? "1" : "0");
|
||||
Dest = json_str(Dest, "SilentLog", Options.EnableLogging ? "0" : "1");
|
||||
if (Options.TSO) {
|
||||
Dest = json_str(Dest, "TSOEnabled", Options.TSO.value() ? "1" : "0");
|
||||
}
|
||||
if (Options.Multiblock) {
|
||||
Dest = json_str(Dest, "Multiblock", Options.Multiblock.value() ? "1" : "0");
|
||||
}
|
||||
|
||||
if (Options.EnableLogging) {
|
||||
Dest = json_str(Dest, "OutputLog", "server");
|
||||
Dest = json_str(Dest, "SilentLog", Options.EnableLogging.value() ? "0" : "1");
|
||||
if (Options.EnableLogging.value()) {
|
||||
Dest = json_str(Dest, "OutputLog", "server");
|
||||
}
|
||||
}
|
||||
Dest = json_objClose(Dest);
|
||||
}
|
||||
|
||||
{
|
||||
if (Options.Thunks_GL || Options.Thunks_Vulkan) {
|
||||
Dest = json_objOpen(Dest, "ThunksDB");
|
||||
Dest = json_str(Dest, "GL", Options.Thunks_GL ? "1" : "0");
|
||||
Dest = json_str(Dest, "Vulkan", Options.Thunks_Vulkan ? "1" : "0");
|
||||
if (Options.Thunks_GL) {
|
||||
Dest = json_str(Dest, "GL", Options.Thunks_GL.value() ? "1" : "0");
|
||||
}
|
||||
if (Options.Thunks_Vulkan) {
|
||||
Dest = json_str(Dest, "Vulkan", Options.Thunks_Vulkan.value() ? "1" : "0");
|
||||
}
|
||||
Dest = json_objClose(Dest);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "DummyHandlers.h"
|
||||
#include "Common/HostFeatures.h"
|
||||
#include "FEXCore/Core/Context.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
@@ -12,6 +13,10 @@
|
||||
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace FEXCore::DiskCache {
|
||||
uint16_t GetFormatVersion();
|
||||
}
|
||||
|
||||
namespace CodeSize {
|
||||
class CodeSizeValidation final {
|
||||
public:
|
||||
@@ -224,6 +229,7 @@ struct TestInfo {
|
||||
|
||||
struct TestHeader {
|
||||
uint64_t Bitness;
|
||||
uint64_t BinaryCacheVersion;
|
||||
uint64_t NumTests {};
|
||||
uint64_t EnabledHostFeatures;
|
||||
uint64_t DisabledHostFeatures;
|
||||
@@ -287,7 +293,34 @@ static bool TestInstructions(FEXCore::Context::Context* CTX, FEXCore::Core::Inte
|
||||
CurrentTest = reinterpret_cast<const TestInfo*>(&CurrentTest->Code[CurrentTest->CodeSize]);
|
||||
}
|
||||
|
||||
auto ExpectedFormatVersion = FEXCore::DiskCache::GetFormatVersion();
|
||||
if (TestHeaderData->BinaryCacheVersion != ExpectedFormatVersion) {
|
||||
// Disk cache binary version updated but failed to update instcount ci tracking.
|
||||
LogMan::Msg::EFmt("Fail: TestHarness binary cache version is '{}' but test json is '{}'", ExpectedFormatVersion,
|
||||
TestHeaderData->BinaryCacheVersion);
|
||||
LogMan::Msg::EFmt("Fail: Please run `ninja instcountci_tests ; ninja instcountci_update_tests` and commit with `git commit -m "
|
||||
"\"InstcountCI: Update\"` to update instcount CI files");
|
||||
TestsPassed = false;
|
||||
}
|
||||
|
||||
if (UpdatedInstructionCountsPath) {
|
||||
if (!TestsPassed && TestHeaderData->BinaryCacheVersion == ExpectedFormatVersion) {
|
||||
// Binary cache versions matched but instructions mismatched. Need to update the format version.
|
||||
// Print a message warning about this otherwise we'll forget about it.
|
||||
LogMan::Msg::EFmt("Fail: Excuse me ma'am, sir, or other unworldly being that is running this software.");
|
||||
LogMan::Msg::EFmt("Fail: InstcountCI results have changed but the FEXCore::DiskCache::FormatVersion hasn't been updated!");
|
||||
LogMan::Msg::EFmt("Fail: This means with your change you are invalidating disk cache entries for everyone. Be sure to know the "
|
||||
"consequences!");
|
||||
LogMan::Msg::EFmt("Fail: Please increment that number, recompile everything and rerun `ninja instcountci_tests ; ninja "
|
||||
"instcountci_update_tests`");
|
||||
LogMan::Msg::EFmt("Fail: DiskCache version should be incremented from '{}' to '{}'", ExpectedFormatVersion, ExpectedFormatVersion + 1);
|
||||
|
||||
// Unlink the file, to ensure it doesn't update with `instcountci_update_tests`
|
||||
unlink(UpdatedInstructionCountsPath);
|
||||
|
||||
return TestsPassed;
|
||||
}
|
||||
|
||||
// Unlink the file.
|
||||
unlink(UpdatedInstructionCountsPath);
|
||||
|
||||
@@ -302,6 +335,12 @@ static bool TestInstructions(FEXCore::Context::Context* CTX, FEXCore::Core::Inte
|
||||
|
||||
FD.Write("{\n", 2);
|
||||
|
||||
FD.Write(fextl::fmt::format("\t\"{}\": {{\n", "Features"));
|
||||
FD.Write(fextl::fmt::format("\t\t\"{}\": {}\n", "BinaryCacheVersion", ExpectedFormatVersion));
|
||||
FD.Write(fextl::fmt::format("\t}},\n"));
|
||||
|
||||
FD.Write(fextl::fmt::format("\t\"{}\": {{\n", "Instructions"));
|
||||
|
||||
CurrentTest = TestsStart;
|
||||
for (size_t i = 0; i < TestHeaderData->NumTests; ++i) {
|
||||
// Get the instruction stats.
|
||||
@@ -330,6 +369,8 @@ static bool TestInstructions(FEXCore::Context::Context* CTX, FEXCore::Core::Inte
|
||||
// Print a null member
|
||||
FD.Write(fextl::fmt::format("\t\"\": \"\""));
|
||||
|
||||
FD.Write(fextl::fmt::format("\t}}\n"));
|
||||
|
||||
FD.Write("}\n", 2);
|
||||
}
|
||||
return TestsPassed;
|
||||
@@ -438,13 +479,9 @@ private:
|
||||
|
||||
class SimpleSyscallHandler : public FEXCore::HLE::SyscallHandler, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
SimpleSyscallHandler() {
|
||||
// Just claim to be linux 64-bit for simplicity.
|
||||
OSABI = FEXCore::HLE::SyscallOSABI::OS_LINUX64;
|
||||
}
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) override {
|
||||
SimpleSyscallHandler() = default;
|
||||
void HandleSyscall(FEXCore::Core::CpuStateFrame* Frame) override {
|
||||
// Don't do anything
|
||||
return 0;
|
||||
}
|
||||
|
||||
// These are no-ops implementations of the SyscallHandler API
|
||||
@@ -660,7 +697,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (!CTX->InitCore()) {
|
||||
return -1;
|
||||
}
|
||||
auto ParentThread = CTX->CreateThread(0, 0);
|
||||
auto ParentThread = CTX->CreateThread();
|
||||
|
||||
// GDT data
|
||||
FEXCore::Core::CPUState::gdt_segment gdt[32] {};
|
||||
|
||||
@@ -11,9 +11,8 @@ namespace FEX::DummyHandlers {
|
||||
|
||||
class DummySyscallHandler : public FEXCore::HLE::SyscallHandler, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) override {
|
||||
void HandleSyscall(FEXCore::Core::CpuStateFrame* Frame) override {
|
||||
// Don't do anything
|
||||
return 0;
|
||||
}
|
||||
|
||||
// These are no-ops implementations of the SyscallHandler API
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <limits.h>
|
||||
|
||||
#include "Common/Config.h"
|
||||
|
||||
namespace FEX {
|
||||
|
||||
@@ -111,7 +111,7 @@ void AOTGenSection(FEXCore::Context::Context* CTX, ELFCodeLoader::LoadedSection&
|
||||
setpriority(PRIO_PROCESS, FHU::Syscalls::gettid(), 19);
|
||||
|
||||
// Setup thread - Each compilation thread uses its own backing FEX thread
|
||||
auto Thread = CTX->CreateThread(0, 0);
|
||||
auto Thread = CTX->CreateThread();
|
||||
fextl::set<uint64_t> ExternalBranchesLocal;
|
||||
CTX->ConfigureAOTGen(Thread, &ExternalBranchesLocal, SectionMaxAddress);
|
||||
|
||||
|
||||
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/DiskCacheFileMapper.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -81,7 +82,9 @@ void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
const auto Style = DisableOutputColors ? fmt::text_style {} : LogMan::DebugLevelStyle(Level);
|
||||
const auto Output = fextl::fmt::format("{} {}\n", fmt::styled(LogMan::DebugLevelStr(Level), Style), Message);
|
||||
write(OutputFD, Output.c_str(), Output.size());
|
||||
fsync(OutputFD);
|
||||
if (Level == LogMan::DebugLevels::ASSERT) {
|
||||
fsync(OutputFD);
|
||||
}
|
||||
}
|
||||
|
||||
void AssertHandler(const char* Message) {
|
||||
@@ -190,6 +193,16 @@ void Shutdown(fextl::vector<FEXCore::Allocator::MemoryRegion>&& MemoryRegions) {
|
||||
}
|
||||
} // namespace FEX::Allocator
|
||||
|
||||
static void* DiskCacheFilemapper(FEXCore::File::File::FileHandleType Handle, uint64_t MapSize) {
|
||||
int mapFD = dup(Handle);
|
||||
void* Result = FEXCore::Allocator::mmap(nullptr, MapSize, PROT_READ, MAP_SHARED | MAP_NORESERVE, mapFD, 0);
|
||||
close(mapFD);
|
||||
if (Result == (void*)-1) {
|
||||
return nullptr;
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
bool InterpreterHandler(fextl::string* Filename, const fextl::string& RootFS, fextl::vector<fextl::string>* args) {
|
||||
int FD {-1};
|
||||
|
||||
@@ -519,6 +532,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
// Setup Thread handlers, so FEXCore can create threads.
|
||||
auto StackTracker = FEX::LinuxEmulation::Threads::SetupThreadHandlers();
|
||||
|
||||
FEXCore::DiskCache::SetFileMapper(DiskCacheFilemapper);
|
||||
|
||||
auto MemoryRegions = FEX::Allocator::InitMemoryRegions(Loader.Is64BitMode());
|
||||
auto Allocator = FEX::Allocator::InitAllocator(Loader.Is64BitMode());
|
||||
|
||||
@@ -571,7 +586,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
// Create a thread without a RIP or stack pointer setup initially.
|
||||
auto ParentThread = SyscallHandler->TM.CreateThread(0, 0);
|
||||
auto ParentThread = SyscallHandler->TM.CreateThread();
|
||||
SyscallHandler->TM.TrackThread(ParentThread);
|
||||
SignalDelegation->RegisterTLSState(ParentThread);
|
||||
ThunkHandler->RegisterTLSState(ParentThread);
|
||||
|
||||
@@ -65,14 +65,11 @@ class AOTSyscallHandler : public FEXCore::HLE::SyscallHandler {
|
||||
class AOTSyscallHandler : public FEXCore::HLE::SyscallHandler, public FEX::HLE::SyscallMmapInterface {
|
||||
#endif
|
||||
public:
|
||||
AOTSyscallHandler(FEXCore::Context::Context& CTX, FEXCore::HLE::SyscallOSABI SyscallOSABI)
|
||||
: CTX(CTX) {
|
||||
OSABI = SyscallOSABI;
|
||||
}
|
||||
AOTSyscallHandler(FEXCore::Context::Context& CTX)
|
||||
: CTX(CTX) {}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) override {
|
||||
void HandleSyscall(FEXCore::Core::CpuStateFrame* Frame) override {
|
||||
// Don't do anything
|
||||
return 0;
|
||||
}
|
||||
|
||||
FEXCore::Context::Context& CTX;
|
||||
@@ -136,6 +133,7 @@ public:
|
||||
#endif
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
static void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
fmt::print("[{}] {}\n", LogMan::DebugLevelStr(Level), Message);
|
||||
}
|
||||
@@ -143,6 +141,7 @@ static void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
static void AssertHandler(const char* Message) {
|
||||
fmt::print("[A] {}\n", Message);
|
||||
}
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
inline bool operator<(const ExecutableFileInfo& a, const ExecutableFileInfo& b) noexcept {
|
||||
@@ -171,7 +170,7 @@ static constexpr size_t DefaultCS {4};
|
||||
#endif
|
||||
|
||||
static FEXCore::Core::InternalThreadState* SetupCompileThread(FEXCore::Context::Context& CTX, bool Is64Bit) {
|
||||
auto Thread = CTX.CreateThread(0, 0);
|
||||
auto Thread = CTX.CreateThread();
|
||||
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &gdt[0];
|
||||
@@ -425,8 +424,7 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
#ifdef _WIN32
|
||||
OvercommitTracker = std::make_unique<FEX::Windows::OvercommitTracker>(IsWine);
|
||||
|
||||
auto SyscallOSABI = FEXCore::HLE::SyscallOSABI::OS_GENERIC;
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX, SyscallOSABI);
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX);
|
||||
|
||||
SyscallHandler->VAFileStart =
|
||||
TryMapImage(SyscallHandler->InvalidationTracker, SyscallHandler->ImageTracker, Binary.FileId, Binary).value_or(0);
|
||||
@@ -439,8 +437,7 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
#else
|
||||
Loader.CalculateHWCaps(CTX.get());
|
||||
|
||||
auto SyscallOSABI = Is64Bit ? FEXCore::HLE::SyscallOSABI::OS_LINUX64 : FEXCore::HLE::SyscallOSABI::OS_LINUX32;
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX, SyscallOSABI);
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX);
|
||||
|
||||
// Populate relocations from ELF file
|
||||
{
|
||||
|
||||
@@ -40,7 +40,7 @@ static bool InitializeSquashFSPipe() {
|
||||
ServerRootFSLockFD = open(RootFSLockFile.c_str(), O_RDWR | O_CLOEXEC, USER_PERMS);
|
||||
if (ServerRootFSLockFD != -1) {
|
||||
// Now that we have opened the file, try to get a write lock.
|
||||
flock lk {
|
||||
struct flock lk {
|
||||
.l_type = F_WRLCK,
|
||||
.l_whence = SEEK_SET,
|
||||
.l_start = 0,
|
||||
@@ -67,7 +67,7 @@ static bool InitializeSquashFSPipe() {
|
||||
return false;
|
||||
} else {
|
||||
// FIFO file was created. Try to get a write lock
|
||||
flock lk {
|
||||
struct flock lk {
|
||||
.l_type = F_WRLCK,
|
||||
.l_whence = SEEK_SET,
|
||||
.l_start = 0,
|
||||
@@ -87,7 +87,7 @@ static bool InitializeSquashFSPipe() {
|
||||
}
|
||||
|
||||
static bool DowngradeRootFSPipeToReadLock() {
|
||||
flock lk {
|
||||
struct flock lk {
|
||||
.l_type = F_RDLCK,
|
||||
.l_whence = SEEK_SET,
|
||||
.l_start = 0,
|
||||
|
||||
@@ -94,6 +94,15 @@ void FileManager::LoadThunkDatabase(fextl::unordered_map<fextl::string, ThunkDBO
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef FEX_STEAM_SUPPORT
|
||||
// Always add pressure-vessel libraries.
|
||||
const auto ArchPrefix = Is64BitMode() ? "x86_64-linux-gnu" : "i386-linux-gnu";
|
||||
PathPrefixes.emplace_back(fextl::fmt::format("/run/gfx/main/usr/lib/{}", ArchPrefix));
|
||||
if (!RootFSIsMultiarch) {
|
||||
PathPrefixes.emplace_back(fextl::fmt::format("/usr/lib/pressure-vessel/overrides/lib/{}", ArchPrefix));
|
||||
}
|
||||
#endif
|
||||
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(FileData, Pool);
|
||||
|
||||
|
||||
@@ -1482,9 +1482,7 @@ static void* ThreadHandler(void* Arg) {
|
||||
}
|
||||
|
||||
void GdbServer::StartThread() {
|
||||
uint64_t OldMask = HLE::ThreadManager::SetSignalMask(~0ULL);
|
||||
gdbServerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
HLE::ThreadManager::SetSignalMask(OldMask);
|
||||
gdbServerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this, {.Internal = true});
|
||||
}
|
||||
|
||||
void GdbServer::OpenListenSocket() {
|
||||
|
||||
@@ -316,15 +316,13 @@ void SeccompEmulator::DeserializeFilters(FEXCore::Core::CpuStateFrame* Frame, in
|
||||
}
|
||||
|
||||
SeccompEmulator::ExecuteFilterResult
|
||||
SeccompEmulator::ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC, FEXCore::HLE::SyscallArguments* Args) {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
if (Thread->Filters.empty()) {
|
||||
SeccompEmulator::ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC, FEX::HLE::SyscallArguments* Args) {
|
||||
if (!HasFilter(Frame)) {
|
||||
// Seccomp not installed. Allow it.
|
||||
return {false, 0};
|
||||
}
|
||||
|
||||
// Reconstruct the RIP from the JITPC.
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
const uint64_t RIP = Thread->Thread->CTX->RestoreRIPFromHostPC(Frame->Thread, JITPC);
|
||||
|
||||
const auto Arch = Is64BitMode() ? AUDIT_ARCH_X86_64 : AUDIT_ARCH_I386;
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include "LinuxSyscalls/ThreadManager.h"
|
||||
|
||||
#include <atomic>
|
||||
#include <csignal>
|
||||
@@ -20,23 +21,16 @@ struct sock_fprog;
|
||||
struct seccomp_data;
|
||||
struct seccomp_notif_sizes;
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Core {
|
||||
struct CpuStateFrame;
|
||||
}
|
||||
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
namespace FEXCore::Core {
|
||||
struct CpuStateFrame;
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEX::HLE {
|
||||
|
||||
class SignalDelegator;
|
||||
class SyscallHandler;
|
||||
struct ThreadStateObject;
|
||||
struct SyscallArguments;
|
||||
|
||||
using SeccompFilterFunc = uint64_t (*)(uint32_t Acc, uint32_t Index, uint32_t Tmp, uint32_t Tmp2, void* Data);
|
||||
struct SeccompFilterInfo final {
|
||||
@@ -65,7 +59,13 @@ public:
|
||||
bool EarlyReturn {};
|
||||
uint64_t Result;
|
||||
};
|
||||
ExecuteFilterResult ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC, FEXCore::HLE::SyscallArguments* Args);
|
||||
ExecuteFilterResult ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC, FEX::HLE::SyscallArguments* Args);
|
||||
|
||||
bool HasFilter(FEXCore::Core::CpuStateFrame* Frame) const {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
return !Thread->Filters.empty();
|
||||
}
|
||||
|
||||
int GetKillSignal() const {
|
||||
return CurrentKillSignal;
|
||||
}
|
||||
|
||||
@@ -808,63 +808,113 @@ uint32_t SyscallHandler::CalculateGuestKernelVersion() {
|
||||
return std::max(LinuxVersion::KernelVersion(5, 15), std::min(LinuxVersion::KernelVersion(6, 11), GetHostKernelVersion()));
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) {
|
||||
// Grab the return address which will be inside the JIT.
|
||||
const uint64_t JITPC = reinterpret_cast<uint64_t>(__builtin_extract_return_addr(__builtin_return_address(0)));
|
||||
template<bool Is64Bit>
|
||||
void SyscallHandler::HandleSyscallImpl(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC) {
|
||||
auto SetResult = [](FEXCore::Core::CpuStateFrame* Frame, uint64_t Result) {
|
||||
const auto Mask = Is64Bit ? ~0ULL : ~0U;
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
const auto SeccompResult = SeccompEmulator.ExecuteFilter(Frame, JITPC, Args);
|
||||
Thread->Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RAX] = Result & Mask;
|
||||
};
|
||||
|
||||
if (SeccompResult.EarlyReturn) {
|
||||
return SeccompResult.Result;
|
||||
if (SeccompEmulator.HasFilter(Frame)) {
|
||||
FEX::HLE::SyscallArguments Args {
|
||||
.Argument =
|
||||
{
|
||||
GetArg(Is64Bit, Frame, 0),
|
||||
GetArg(Is64Bit, Frame, 1),
|
||||
GetArg(Is64Bit, Frame, 2),
|
||||
GetArg(Is64Bit, Frame, 3),
|
||||
GetArg(Is64Bit, Frame, 4),
|
||||
GetArg(Is64Bit, Frame, 5),
|
||||
GetArg(Is64Bit, Frame, 6),
|
||||
},
|
||||
};
|
||||
const auto SeccompResult = SeccompEmulator.ExecuteFilter(Frame, JITPC, &Args);
|
||||
|
||||
if (SeccompResult.EarlyReturn) {
|
||||
SetResult(Frame, SeccompResult.Result);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Args->Argument[0] >= Definitions.size()) {
|
||||
return -ENOSYS;
|
||||
const auto SyscallNum = GetArg(Is64Bit, Frame, 0);
|
||||
if (SyscallNum >= Definitions.size()) {
|
||||
SetResult(Frame, -ENOSYS);
|
||||
return;
|
||||
}
|
||||
|
||||
auto& Def = Definitions[Args->Argument[0]];
|
||||
auto& Def = Definitions[SyscallNum];
|
||||
uint64_t Result {};
|
||||
switch (Def.NumArgs) {
|
||||
case 0: Result = std::invoke(Def.Ptr0, Frame); break;
|
||||
case 1: Result = std::invoke(Def.Ptr1, Frame, Args->Argument[1]); break;
|
||||
case 2: Result = std::invoke(Def.Ptr2, Frame, Args->Argument[1], Args->Argument[2]); break;
|
||||
case 3: Result = std::invoke(Def.Ptr3, Frame, Args->Argument[1], Args->Argument[2], Args->Argument[3]); break;
|
||||
case 4: Result = std::invoke(Def.Ptr4, Frame, Args->Argument[1], Args->Argument[2], Args->Argument[3], Args->Argument[4]); break;
|
||||
case 1: Result = std::invoke(Def.Ptr1, Frame, GetArg(Is64Bit, Frame, 1)); break;
|
||||
case 2: Result = std::invoke(Def.Ptr2, Frame, GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2)); break;
|
||||
case 3: Result = std::invoke(Def.Ptr3, Frame, GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3)); break;
|
||||
case 4:
|
||||
Result =
|
||||
std::invoke(Def.Ptr4, Frame, GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3), GetArg(Is64Bit, Frame, 4));
|
||||
break;
|
||||
case 5:
|
||||
Result = std::invoke(Def.Ptr5, Frame, Args->Argument[1], Args->Argument[2], Args->Argument[3], Args->Argument[4], Args->Argument[5]);
|
||||
Result = std::invoke(Def.Ptr5, Frame, GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3),
|
||||
GetArg(Is64Bit, Frame, 4), GetArg(Is64Bit, Frame, 5));
|
||||
break;
|
||||
case 6:
|
||||
Result = std::invoke(Def.Ptr6, Frame, Args->Argument[1], Args->Argument[2], Args->Argument[3], Args->Argument[4], Args->Argument[5],
|
||||
Args->Argument[6]);
|
||||
Result = std::invoke(Def.Ptr6, Frame, GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3),
|
||||
GetArg(Is64Bit, Frame, 4), GetArg(Is64Bit, Frame, 5), GetArg(Is64Bit, Frame, 6));
|
||||
break;
|
||||
// for missing syscalls
|
||||
case 255: return std::invoke(Def.Ptr1, Frame, Args->Argument[0]);
|
||||
case 255: Result = std::invoke(Def.Ptr1, Frame, GetArg(Is64Bit, Frame, 0)); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled syscall: {}", Args->Argument[0]);
|
||||
return -1;
|
||||
LOGMAN_MSG_A_FMT("Unhandled syscall: {}", GetArg(Is64Bit, Frame, 0));
|
||||
Result = -ENOSYS;
|
||||
break;
|
||||
}
|
||||
#ifdef DEBUG_STRACE
|
||||
Strace(Args, Result);
|
||||
Strace(Frame, Result);
|
||||
#endif
|
||||
return Result;
|
||||
SetResult(Frame, Result);
|
||||
}
|
||||
|
||||
void SyscallHandler::HandleSyscall(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
// Grab the return address which will be inside the JIT.
|
||||
const uint64_t JITPC = reinterpret_cast<uint64_t>(__builtin_extract_return_addr(__builtin_return_address(0)));
|
||||
const auto Is64Bit = Is64BitMode();
|
||||
|
||||
// TODO: At some point these will be runtime selectable based on `syscall` versus `int 0x80` entrypoint.
|
||||
if (Is64Bit) {
|
||||
HandleSyscallImpl<true>(Frame, JITPC);
|
||||
} else {
|
||||
HandleSyscallImpl<false>(Frame, JITPC);
|
||||
}
|
||||
|
||||
// Skip past the `syscall` or `int 0x80` instruction. Both of which are 2-bytes.
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
Thread->Thread->CurrentFrame->State.rip += 2;
|
||||
}
|
||||
|
||||
#ifdef DEBUG_STRACE
|
||||
void SyscallHandler::Strace(FEXCore::HLE::SyscallArguments* Args, uint64_t Ret) {
|
||||
auto& Def = Definitions[Args->Argument[0]];
|
||||
void SyscallHandler::Strace(FEXCore::Core::CpuStateFrame* Frame, uint64_t Ret) {
|
||||
const auto Is64Bit = Is64BitMode();
|
||||
auto& Def = Definitions[GetArg(Is64Bit, Frame, 0)];
|
||||
switch (Def.NumArgs) {
|
||||
case 0: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Ret); break;
|
||||
case 1: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Args->Argument[1], Ret); break;
|
||||
case 2: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Args->Argument[1], Args->Argument[2], Ret); break;
|
||||
case 3: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Args->Argument[1], Args->Argument[2], Args->Argument[3], Ret); break;
|
||||
case 4: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Args->Argument[1], Args->Argument[2], Args->Argument[3], Args->Argument[4], Ret); break;
|
||||
case 1: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), GetArg(Is64Bit, Frame, 1), Ret); break;
|
||||
case 2: LogMan::Msg::DFmt(Def.StraceFmt.c_str(), GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), Ret); break;
|
||||
case 3:
|
||||
LogMan::Msg::DFmt(Def.StraceFmt.c_str(), GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3), Ret);
|
||||
break;
|
||||
case 4:
|
||||
LogMan::Msg::DFmt(Def.StraceFmt.c_str(), GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3),
|
||||
GetArg(Is64Bit, Frame, 4), Ret);
|
||||
break;
|
||||
case 5:
|
||||
LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Args->Argument[1], Args->Argument[2], Args->Argument[3], Args->Argument[4], Args->Argument[5], Ret);
|
||||
LogMan::Msg::DFmt(Def.StraceFmt.c_str(), GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3),
|
||||
GetArg(Is64Bit, Frame, 4), GetArg(Is64Bit, Frame, 5), Ret);
|
||||
break;
|
||||
case 6:
|
||||
LogMan::Msg::DFmt(Def.StraceFmt.c_str(), Args->Argument[1], Args->Argument[2], Args->Argument[3], Args->Argument[4], Args->Argument[5],
|
||||
Args->Argument[6], Ret);
|
||||
LogMan::Msg::DFmt(Def.StraceFmt.c_str(), GetArg(Is64Bit, Frame, 1), GetArg(Is64Bit, Frame, 2), GetArg(Is64Bit, Frame, 3),
|
||||
GetArg(Is64Bit, Frame, 4), GetArg(Is64Bit, Frame, 5), GetArg(Is64Bit, Frame, 6), Ret);
|
||||
break;
|
||||
default: break;
|
||||
}
|
||||
|
||||
Loaded 100 of 255 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user