mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 19:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d1a4029bc5 | ||
|
|
97070aad25 | ||
|
|
39640185a3 | ||
|
|
4b17506ffe | ||
|
|
2435ebecbe | ||
|
|
410a35b968 | ||
|
|
c8d234a767 | ||
|
|
2bf87ff40f | ||
|
|
ba347c49c9 | ||
|
|
81434cd233 | ||
|
|
46d0df9cba | ||
|
|
b7f58e68c5 | ||
|
|
6ba2accbdc | ||
|
|
29ee94d331 | ||
|
|
cdae654fe4 | ||
|
|
a506c84bc7 | ||
|
|
8e4a47181b | ||
|
|
b5241e0f60 | ||
|
|
57ed466a7f | ||
|
|
4f46f55f2d | ||
|
|
69cfc78ee1 | ||
|
|
04f1ab8571 | ||
|
|
95694b2017 | ||
|
|
00bed2f0c0 | ||
|
|
e718fc35f8 | ||
|
|
717015bae8 | ||
|
|
530d3d809b | ||
|
|
596b32d15c | ||
|
|
8d3918b4f0 | ||
|
|
4c7e31513b | ||
|
|
765509d7f5 | ||
|
|
dbb58d10a6 | ||
|
|
d448976b3c | ||
|
|
de6931b1f5 | ||
|
|
35268d185e | ||
|
|
beef9eee0a | ||
|
|
34a274d4e6 | ||
|
|
ef28a6c19a | ||
|
|
50b5971ee5 | ||
|
|
9a70ae18ea | ||
|
|
d10853b775 | ||
|
|
cdf6a16efc | ||
|
|
02d7261f51 | ||
|
|
9c9ddeffbe | ||
|
|
afa8b3a5c9 | ||
|
|
bbcd4c168c | ||
|
|
d22bd9cac7 | ||
|
|
5ccf25196e | ||
|
|
caf15a2dac | ||
|
|
982a05450c | ||
|
|
3e381b742c | ||
|
|
6b82664166 | ||
|
|
bb30a2eb1e | ||
|
|
b3fdf5c48f | ||
|
|
17d5ed847f | ||
|
|
4335d17fc0 | ||
|
|
ae07958577 | ||
|
|
c51b9ba3d6 | ||
|
|
cc6ff5e9e6 | ||
|
|
b968ea7e7e | ||
|
|
d14b6e160e | ||
|
|
b09b9488ef | ||
|
|
a8120ee7ef | ||
|
|
fb82059750 | ||
|
|
0fc6240d72 | ||
|
|
a7c6fdb1fc | ||
|
|
b76a2963cf | ||
|
|
42b0fbd34c | ||
|
|
f69ef8606f | ||
|
|
afa5ad5f9f | ||
|
|
1fc82708e9 | ||
|
|
917cbbadde | ||
|
|
c37dc81839 | ||
|
|
df718d55ef | ||
|
|
54412f1d5e | ||
|
|
da76023bea | ||
|
|
41e9309a36 | ||
|
|
116268b275 | ||
|
|
73802492b8 | ||
|
|
7cd4d53fa9 | ||
|
|
c44757975e | ||
|
|
f25cdcdf63 | ||
|
|
5d37253e85 | ||
|
|
cd5f42ec79 | ||
|
|
6651f9e94b | ||
|
|
e3ee579f92 | ||
|
|
73e7240574 | ||
|
|
391f9aa97d | ||
|
|
8ad54e7bd5 | ||
|
|
9eccc01dd3 | ||
|
|
39c1f816fc | ||
|
|
a32b892787 | ||
|
|
1b144ba3f0 | ||
|
|
6a39a8db72 | ||
|
|
602c530615 | ||
|
|
c8c27f26f7 | ||
|
|
549cdc4c2c | ||
|
|
3160e0a430 | ||
|
|
dcebe85f3a | ||
|
|
2ba0b66426 | ||
|
|
5c9543f159 | ||
|
|
f6e3689f30 | ||
|
|
a761343717 | ||
|
|
b4c47a3d24 | ||
|
|
906988c49b | ||
|
|
4186b2ad82 | ||
|
|
2943cff73f | ||
|
|
53ac5579fb | ||
|
|
0d7a9f911a | ||
|
|
dce9de222d | ||
|
|
d46722a95c | ||
|
|
3ba4da7736 | ||
|
|
6abf5b90b7 | ||
|
|
00aa4ddea0 | ||
|
|
d0c6f9de22 | ||
|
|
a85cc85081 | ||
|
|
75793300f2 | ||
|
|
672805584e | ||
|
|
5b4fd590d1 | ||
|
|
0ccd38f593 | ||
|
|
b46e5d4488 | ||
|
|
d7223d598f | ||
|
|
7a0368132d | ||
|
|
aff3914a66 | ||
|
|
8876047875 | ||
|
|
78e2aa16f0 | ||
|
|
3ef695cf70 | ||
|
|
c4d8dd6413 | ||
|
|
a49d30f6e2 | ||
|
|
d80daf2692 | ||
|
|
0e5c9e8b06 | ||
|
|
6bc4aed82c | ||
|
|
e69e1200f5 | ||
|
|
81e253b06b | ||
|
|
43dcc84c07 | ||
|
|
6e9d5f00de | ||
|
|
a65ca9663f | ||
|
|
02b767c0ea | ||
|
|
4b1c1d266d | ||
|
|
d9bf140971 | ||
|
|
1402776ba6 | ||
|
|
e36fb47d98 | ||
|
|
2bb37357c0 | ||
|
|
7494ac7615 | ||
|
|
44bc3fb90b | ||
|
|
423e29ba42 |
No files matched your search
@@ -47,3 +47,6 @@
|
||||
[submodule "External/jemalloc_glibc"]
|
||||
path = External/jemalloc_glibc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
[submodule "External/tracy"]
|
||||
path = External/tracy
|
||||
url = https://github.com/wolfpld/tracy
|
||||
+21
-1
@@ -30,7 +30,7 @@ option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only u
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
@@ -61,6 +61,22 @@ if (ENABLE_FEXCORE_PROFILER)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
elseif (FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=2)
|
||||
add_definitions(-DTRACY_ENABLE=1)
|
||||
# Required so that Tracy will only start in the selected guest application
|
||||
add_definitions(-DTRACY_MANUAL_LIFETIME=1)
|
||||
add_definitions(-DTRACY_DELAYED_INIT=1)
|
||||
# This interferes with FEX's signal handling
|
||||
add_definitions(-DTRACY_NO_CRASH_HANDLER=1)
|
||||
# Tracy can gather call stack samples in regular intervals, but this
|
||||
# isn't useful for us since it would usually sample opaque JIT code
|
||||
add_definitions(-DTRACY_NO_SAMPLING=1)
|
||||
# This pulls in libbacktrace which allocators in global constructors (before FEX can set up its allocator hooks)
|
||||
add_definitions(-DTRACY_NO_CALLSTACK=1)
|
||||
if (MINGW_BUILD)
|
||||
message(FATAL_ERROR "Tracy profiler not supported")
|
||||
endif()
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
@@ -270,6 +286,10 @@ if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_subdirectory(External/tracy)
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
|
||||
|
||||
@@ -95,10 +95,6 @@ public:
|
||||
|
||||
protected:
|
||||
|
||||
void ResetBuffer() {
|
||||
CurrentOffset = BufferBase;
|
||||
}
|
||||
|
||||
uint8_t* BufferBase;
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
|
||||
@@ -767,6 +767,7 @@ public:
|
||||
void st1(ARMEmitter::SubRegSize size, T rt, uint32_t Index, ARMEmitter::Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == SubRegSizeInBits(size), "Post-Index size must match element size");
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
uint32_t Q;
|
||||
uint32_t R = 0;
|
||||
@@ -808,6 +809,7 @@ public:
|
||||
void ld1(ARMEmitter::SubRegSize size, T rt, uint32_t Index, ARMEmitter::Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == SubRegSizeInBits(size), "Post-Index size must match element size");
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
uint32_t Q;
|
||||
uint32_t R = 0;
|
||||
@@ -899,6 +901,7 @@ public:
|
||||
void st2(SubRegSize size, T rt, T rt2, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 2), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2), "rt and rt2 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -942,6 +945,7 @@ public:
|
||||
void ld2(SubRegSize size, T rt, T rt2, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 2), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2), "rt and rt2 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -985,6 +989,7 @@ public:
|
||||
void st3(SubRegSize size, T rt, T rt2, T rt3, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 3), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3), "rt, rt2, and rt3 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -1028,6 +1033,7 @@ public:
|
||||
void ld3(SubRegSize size, T rt, T rt2, T rt3, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 3), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3), "rt, rt2, and rt3 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -1071,6 +1077,7 @@ public:
|
||||
void st4(SubRegSize size, T rt, T rt2, T rt3, T rt4, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 4), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3, rt4), "rt, rt2, rt3, and rt4 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -1114,6 +1121,7 @@ public:
|
||||
void ld4(SubRegSize size, T rt, T rt2, T rt3, T rt4, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 4), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3, rt4), "rt, rt2, rt3, and rt4 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
|
||||
@@ -2379,13 +2379,13 @@ public:
|
||||
} else if (srcsize == SubRegSize::i32Bit) {
|
||||
// Srcsize = fp32, opc1 encodes dst size
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b10;
|
||||
opc2 = 0b10;
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b10 : 0b00;
|
||||
} else if (srcsize == SubRegSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
// SrcSize = fp64, opc2 encodes dst size
|
||||
opc1 = 0b11;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b00 : 0b00;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b00;
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -2400,13 +2400,13 @@ public:
|
||||
} else if (srcsize == SubRegSize::i32Bit) {
|
||||
// Srcsize = fp32, opc1 encodes dst size
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b10;
|
||||
opc2 = 0b10;
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b10 : 0b00;
|
||||
} else if (srcsize == SubRegSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
// SrcSize = fp64, opc2 encodes dst size
|
||||
opc1 = 0b11;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b00 : 0b00;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b00;
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -5029,7 +5029,7 @@ private:
|
||||
void SVE2IntegerMultiplyLong(uint32_t SUT, SubRegSize size, ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
// PMULLB and PMULLT support the use of 128-bit element sizes (with the SVE2PMULL128 extension)
|
||||
if (SUT == 0b010 || SUT == 0b011) {
|
||||
LOGMAN_THROW_A_FMT(size != SubRegSize::i8Bit, "Can't use 8-bit element size");
|
||||
LOGMAN_THROW_A_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i32Bit, "Can't use 8-bit or 32-bit element size");
|
||||
|
||||
// 128-bit variant is encoded as if it were 8-bit (0b00)
|
||||
if (size == SubRegSize::i128Bit) {
|
||||
|
||||
+1
-136
@@ -9,25 +9,6 @@
|
||||
"@PREFIX_LIB@/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
"Library": "libGLESv2-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libGLESv2.so",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libX11.so",
|
||||
"@PREFIX_LIB@/libX11.so.6",
|
||||
"@PREFIX_LIB@/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Overlay": [
|
||||
@@ -36,89 +17,6 @@
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb.so",
|
||||
"@PREFIX_LIB@/libxcb.so.1",
|
||||
"@PREFIX_LIB@/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-present.so",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxshmfence.so",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
@@ -141,38 +39,6 @@
|
||||
"@PREFIX_LIB@/libfex_thunk_test.so"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXrender.so",
|
||||
"@PREFIX_LIB@/libXrender.so.1",
|
||||
"@PREFIX_LIB@/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXext.so",
|
||||
"@PREFIX_LIB@/libXext.so.6",
|
||||
"@PREFIX_LIB@/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXfixes.so",
|
||||
"@PREFIX_LIB@/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libOpenCL.so",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
@@ -180,7 +46,6 @@
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
}
|
||||
+1
Submodule External/tracy added at 5d542dc09f.
Vendored
+1
-1
Submodule External/vixl updated: 3180ab603b...84bc10c107.
@@ -188,27 +188,33 @@ def print_man_environment_tail():
|
||||
|
||||
# Additional environment variables that live outside of the normal loop
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG_LOCATION",
|
||||
"APP_CONFIG_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for configuration files",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG",
|
||||
"APP_CONFIG",
|
||||
[
|
||||
"Allows the user to override where FEX looks for only the application config file",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
|
||||
"This will override this file location",
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_DATA_LOCATION",
|
||||
"APP_DATA_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for data files",
|
||||
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
|
||||
@@ -218,7 +224,7 @@ def print_man_environment_tail():
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
"PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
|
||||
@@ -105,6 +105,7 @@ set (SRCS
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/StringCompareFallbacks.cpp
|
||||
Interface/Core/JIT/JIT.cpp
|
||||
Interface/Core/JIT/ALUOps.cpp
|
||||
Interface/Core/JIT/AtomicOps.cpp
|
||||
@@ -337,6 +338,10 @@ add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
target_link_libraries(FEXCore_Base TracyClient)
|
||||
endif()
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "Common/VectorRegType.h"
|
||||
|
||||
extern "C" {
|
||||
#include "SoftFloat-3e/platform.h"
|
||||
#include "SoftFloat-3e/softfloat.h"
|
||||
@@ -476,6 +478,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
FEXCore::VectorRegType Ret {};
|
||||
memcpy(&Ret, this, sizeof(*this));
|
||||
return Ret;
|
||||
}
|
||||
|
||||
LIBRARY_PRECISION ToFMax(softfloat_state* state) const {
|
||||
#ifdef _WIN32
|
||||
return ToF64(state);
|
||||
@@ -567,12 +575,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
*this = i32_to_extF80(rhs);
|
||||
}
|
||||
|
||||
X80SoftFloat(const FEXCore::VectorRegType rhs) {
|
||||
memcpy(this, &rhs, sizeof(*this));
|
||||
}
|
||||
|
||||
void operator=(extFloat80_t rhs) {
|
||||
Significand = rhs.signif;
|
||||
Exponent = rhs.signExp & 0x7FFF;
|
||||
Sign = rhs.signExp >> 15;
|
||||
}
|
||||
|
||||
operator FEXCore::VectorRegType() const {
|
||||
return ToVector();
|
||||
}
|
||||
|
||||
operator extFloat80_t() const {
|
||||
extFloat80_t Result {};
|
||||
Result.signif = Significand;
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
#ifdef _M_ARM_64
|
||||
// Can't use uint8x16_t directly from arm_neon.h here.
|
||||
// Overrides softfloat-3e's defines which causes problems.
|
||||
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
|
||||
#elif defined(_M_X86_64)
|
||||
using VectorRegType = __m128i;
|
||||
#endif
|
||||
} // namespace FEXCore
|
||||
@@ -3,7 +3,7 @@
|
||||
"CPU": {
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
@@ -363,6 +363,14 @@
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
|
||||
]
|
||||
},
|
||||
"ProfileStats": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
|
||||
@@ -248,8 +248,6 @@ public:
|
||||
~ContextImpl();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const BlockDelinkerFunc& delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
|
||||
@@ -360,8 +358,6 @@ private:
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
|
||||
@@ -48,6 +48,10 @@ namespace CPU {
|
||||
{0x8000'0000'8000'0000ULL, 0x8000'0000'8000'0000ULL}, // NAMED_VECTOR_CVTMAX_I32
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_I64
|
||||
{0x0000'0000'0000'0000ULL, 0x0000'0000'0000'8000ULL}, // NAMED_VECTOR_F80_SIGN_MASK
|
||||
{0x5A82'7999'5A82'7999ULL, 0x5A82'7999'5A82'7999ULL}, // NAMED_VECTOR_SHA1RNDS_K0
|
||||
{0x6ED9'EBA1'6ED9'EBA1ULL, 0x6ED9'EBA1'6ED9'EBA1ULL}, // NAMED_VECTOR_SHA1RNDS_K1
|
||||
{0x8F1B'BCDC'8F1B'BCDCULL, 0x8F1B'BCDC'8F1B'BCDCULL}, // NAMED_VECTOR_SHA1RNDS_K2
|
||||
{0xCA62'C1D6'CA62'C1D6ULL, 0xCA62'C1D6'CA62'C1D6ULL}, // NAMED_VECTOR_SHA1RNDS_K3
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
|
||||
@@ -87,9 +87,6 @@ namespace CPU {
|
||||
// The length of the guest code for this block.
|
||||
size_t GuestSize;
|
||||
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
@@ -99,7 +96,10 @@ namespace CPU {
|
||||
// Shared-code modification spin-loop futex.
|
||||
uint32_t SpinLockFutex;
|
||||
|
||||
uint32_t _Pad;
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
uint8_t _Pad[3];
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -473,6 +473,7 @@ void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
Profiler::PostForkAction(Child);
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
@@ -496,10 +497,6 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
@@ -773,8 +770,9 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedJITTime);
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
@@ -843,7 +841,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Insert to lookup cache
|
||||
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr);
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
@@ -907,19 +905,10 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
|
||||
@@ -607,7 +607,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
uint16_t LocalOp = OPD(Info->Type, PrefixType, ModRM.reg);
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
#undef OPD
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM) {
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM && ModRM.mod == 0b11) {
|
||||
// Everything in this group is privileged instructions aside from XGETBV
|
||||
constexpr std::array<uint8_t, 8> RegToField = {
|
||||
255, 0, 1, 2, 255, 255, 255, 3,
|
||||
|
||||
@@ -5,6 +5,9 @@
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
softfloat_state State {};
|
||||
@@ -35,12 +38,14 @@ FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, float src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, float src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t FCW, double src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
@@ -48,7 +53,8 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
|
||||
bool eq, lt, nan;
|
||||
@@ -70,37 +76,43 @@ struct OpHandlers<IR::OP_F80CMP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF32(&State);
|
||||
return X80SoftFloat(src).ToF32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF64(&State);
|
||||
return X80SoftFloat(src).ToF64(&State);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI16(&State);
|
||||
return X80SoftFloat(src).ToI16(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI32(&State);
|
||||
return X80SoftFloat(src).ToI32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI64(&State);
|
||||
return X80SoftFloat(src).ToI64(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
auto rv = extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
auto rv = extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX || rv < INT16_MIN) {
|
||||
///< Indefinite value for 16-bit conversions.
|
||||
@@ -110,31 +122,36 @@ struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
return extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i64(&State, src, softfloat_round_minMag, false);
|
||||
return extF80_to_i64(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t FCW, int16_t src) {
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle2(uint16_t FCW, int16_t src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat(src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, int32_t src) {
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, int32_t src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FRNDINT(&State, Src1);
|
||||
}
|
||||
@@ -142,7 +159,8 @@ struct OpHandlers<IR::OP_F80ROUND> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::F2XM1(&State, Src1);
|
||||
}
|
||||
@@ -150,7 +168,8 @@ struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FTAN(&State, Src1);
|
||||
}
|
||||
@@ -158,7 +177,8 @@ struct OpHandlers<IR::OP_F80TAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSQRT(&State, Src1);
|
||||
}
|
||||
@@ -166,7 +186,8 @@ struct OpHandlers<IR::OP_F80SQRT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSIN(&State, Src1);
|
||||
}
|
||||
@@ -174,7 +195,8 @@ struct OpHandlers<IR::OP_F80SIN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FCOS(&State, Src1);
|
||||
}
|
||||
@@ -182,21 +204,24 @@ struct OpHandlers<IR::OP_F80COS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FADD(&State, Src1, Src2);
|
||||
}
|
||||
@@ -204,7 +229,8 @@ struct OpHandlers<IR::OP_F80ADD> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSUB(&State, Src1, Src2);
|
||||
}
|
||||
@@ -212,7 +238,8 @@ struct OpHandlers<IR::OP_F80SUB> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FMUL(&State, Src1, Src2);
|
||||
}
|
||||
@@ -220,7 +247,8 @@ struct OpHandlers<IR::OP_F80MUL> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FDIV(&State, Src1, Src2);
|
||||
}
|
||||
@@ -228,7 +256,8 @@ struct OpHandlers<IR::OP_F80DIV> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FYL2X(&State, Src1, Src2);
|
||||
}
|
||||
@@ -236,7 +265,8 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FATAN(&State, Src1, Src2);
|
||||
}
|
||||
@@ -244,7 +274,8 @@ struct OpHandlers<IR::OP_F80ATAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM1(&State, Src1, Src2);
|
||||
}
|
||||
@@ -252,7 +283,8 @@ struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM(&State, Src1, Src2);
|
||||
}
|
||||
@@ -260,7 +292,8 @@ struct OpHandlers<IR::OP_F80FPREM> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSCALE(&State, Src1, Src2);
|
||||
}
|
||||
@@ -268,63 +301,72 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
}
|
||||
@@ -335,7 +377,9 @@ struct OpHandlers<IR::OP_F64SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1q, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
X80SoftFloat Src1 = Src1q;
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
@@ -376,7 +420,8 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
|
||||
uint64_t BCD {};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
@@ -15,13 +16,14 @@ static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHan
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex, false};
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, FEXCore::Core::CpuStateFrame*), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_PTR, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex, false};
|
||||
FallbackInfo
|
||||
GetFallbackInfo(double (*fn)(uint16_t, double, double, FEXCore::Core::CpuStateFrame*), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64_PTR, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
@@ -86,11 +88,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_F32_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_F64_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -100,11 +102,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -117,28 +119,31 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
*Info = {FABI_I16_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I16_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
*Info = {FABI_I32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
*Info = {FABI_I64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -147,7 +152,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
*Info = {FABI_I64_I16_F80_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
(Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP), SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -157,11 +162,13 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i16Bit: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_I16_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_I32_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -169,16 +176,16 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
@@ -229,7 +236,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I32_V128_V128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
|
||||
default: break;
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include <arm_neon.h>
|
||||
#endif
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#ifdef _M_ARM_64
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
if (is_using_words) {
|
||||
uint16x8_t a = vreinterpretq_u16_u8(data);
|
||||
uint16x8_t VIndexes {};
|
||||
const uint16x8_t VIndex16 = vdupq_n_u16(8);
|
||||
uint16_t Indexes[8] = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7,
|
||||
};
|
||||
memcpy(&VIndexes, Indexes, sizeof(VIndexes));
|
||||
auto MaskResult = vceqzq_u16(a);
|
||||
auto SelectResult = vbslq_u16(MaskResult, VIndexes, VIndex16);
|
||||
return vminvq_u16(SelectResult);
|
||||
} else {
|
||||
uint8x16_t VIndexes {};
|
||||
const uint8x16_t VIndex16 = vdupq_n_u8(16);
|
||||
uint8_t Indexes[16] = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
|
||||
};
|
||||
memcpy(&VIndexes, Indexes, sizeof(VIndexes));
|
||||
auto MaskResult = vceqzq_u8(data);
|
||||
auto SelectResult = vbslq_u8(MaskResult, VIndexes, VIndex16);
|
||||
return vminvq_u8(SelectResult);
|
||||
}
|
||||
}
|
||||
#else
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR uint32_t OpHandlers<IR::OP_VPCMPISTRX>::handle(FEXCore::VectorRegType lhs, FEXCore::VectorRegType rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
__uint128_t lhs_i;
|
||||
memcpy(&lhs_i, &lhs, sizeof(lhs_i));
|
||||
__uint128_t rhs_i;
|
||||
memcpy(&rhs_i, &rhs, sizeof(rhs_i));
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs_i, valid_lhs, rhs_i, valid_rhs, control);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -6,9 +6,9 @@
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Common/VectorRegType.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -344,51 +344,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(VectorRegType lhs, VectorRegType rhs, uint16_t control);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,8 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -16,22 +14,22 @@ struct IROp_Header;
|
||||
namespace FEXCore::CPU {
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_F80_I16_F32,
|
||||
FABI_F80_I16_F64,
|
||||
FABI_F80_I16_I16,
|
||||
FABI_F80_I16_I32,
|
||||
FABI_F32_I16_F80,
|
||||
FABI_F64_I16_F80,
|
||||
FABI_F64_I16_F64,
|
||||
FABI_F64_I16_F64_F64,
|
||||
FABI_I16_I16_F80,
|
||||
FABI_I32_I16_F80,
|
||||
FABI_I64_I16_F80,
|
||||
FABI_I64_I16_F80_F80,
|
||||
FABI_F80_I16_F80,
|
||||
FABI_F80_I16_F80_F80,
|
||||
FABI_F80_I16_F32_PTR,
|
||||
FABI_F80_I16_F64_PTR,
|
||||
FABI_F80_I16_I16_PTR,
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_F80_PTR,
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
|
||||
@@ -169,6 +169,85 @@ DEF_OP(VSha1H) {
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
|
||||
DEF_OP(VSha1C) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1C>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1c(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1M) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1M>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1m(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1P) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1P>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1p(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1SU1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1SU1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1su1(Dst, Src2);
|
||||
} else if (Dst != Src2) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1su1(Dst, Src2);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1su1(VTMP1, Src2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
@@ -185,6 +264,23 @@ DEF_OP(VSha256U0) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst != Src1 && Dst != Src1) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
sha256su1(Dst, Src1, Src2);
|
||||
} else {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
sha256su1(VTMP1, Src1, Src2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -11,6 +11,7 @@ desc: Main glue logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
@@ -87,16 +88,13 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
auto FillF80Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
mov(TMP2, ARMEmitter::XReg::x1);
|
||||
mov(VTMP1.Q(), ARMEmitter::VReg::v0.Q());
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, TMP2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
};
|
||||
|
||||
auto FillF64Result = [&]() {
|
||||
@@ -120,52 +118,16 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
};
|
||||
|
||||
switch (Info.ABI) {
|
||||
case FABI_F80_I16_F32: {
|
||||
case FABI_F80_I16_F32_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, float, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
@@ -173,20 +135,59 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F32_I16_F80: {
|
||||
case FABI_F80_I16_F64_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_I16_PTR:
|
||||
case FABI_F80_I16_I32_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16_PTR) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x2, STATE);
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, uint32_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F32_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -198,43 +199,45 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
fmov(Dst.S(), VTMP1.S());
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F80: {
|
||||
case FABI_F64_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64: {
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
@@ -249,30 +252,31 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_I16_I16_F80: {
|
||||
case FABI_I16_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -283,38 +287,38 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I32_I16_F80: {
|
||||
case FABI_I32_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I64_I16_F80: {
|
||||
case FABI_I64_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -325,24 +329,28 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I64_I16_F80_F80: {
|
||||
case FABI_I64_I16_F80_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -353,42 +361,47 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_F80_I16_F80: {
|
||||
case FABI_F80_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
case FABI_F80_I16_F80_F80: {
|
||||
case FABI_F80_I16_F80_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint64_t>(
|
||||
ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
@@ -430,7 +443,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
case FABI_I32_V128_V128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
@@ -439,19 +452,22 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint16_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
@@ -466,7 +482,6 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
|
||||
@@ -759,7 +759,8 @@ DEF_OP(LoadMemTSO) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -1547,7 +1548,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
case IR::OpSize::i16Bit: strh(Src, MemSrc); break;
|
||||
@@ -1658,8 +1659,8 @@ DEF_OP(StoreMemPair) {
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetReg(Op->Value1.ID());
|
||||
const auto Src2 = GetReg(Op->Value2.ID());
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), Addr, Op->Offset); break;
|
||||
case IR::OpSize::i64Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), Addr, Op->Offset); break;
|
||||
@@ -1691,10 +1692,11 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -1711,7 +1713,7 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -1763,7 +1765,7 @@ DEF_OP(MemSet) {
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Value = GetZeroableReg(Op->Value);
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
@@ -2312,7 +2314,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
@@ -2332,7 +2334,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
|
||||
@@ -1356,13 +1356,14 @@ DEF_OP(VFMin) {
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
// NOTE: We don't directly use FMIN here for any of the implementations,
|
||||
// NOTE: We don't directly use FMIN** here for any of the implementations,
|
||||
// because it has undesirable NaN handling behavior (it sets
|
||||
// entries either to the incoming NaN value*, or the default NaN
|
||||
// depending on FPCR flags set). We want behavior that sets NaN
|
||||
// entries to zero for the comparison result.
|
||||
//
|
||||
// * - Not exactly (differs slightly with SNaNs), but close enough for the explanation
|
||||
// ** - Unless the host supports AFP.AH, which allows FMIN/FMAX to select the second source element as expected of x86.
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
@@ -1391,6 +1392,12 @@ DEF_OP(VFMin) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!IsScalar, "should use VFMinScalarInsert instead");
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
// AFP.AH lets fmin behave like x86 min
|
||||
fmin(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
return;
|
||||
}
|
||||
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on false.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
@@ -1443,6 +1450,12 @@ DEF_OP(VFMax) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!IsScalar, "should use VFMaxScalarInsert instead");
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
// AFP.AH lets fmax behave like x86 max
|
||||
fmax(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
return;
|
||||
}
|
||||
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on true.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
@@ -1507,7 +1520,10 @@ DEF_OP(VFRecp) {
|
||||
fdiv(Dst.D(), VTMP1.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unexpected ElementSize for {}", __func__);
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
|
||||
@@ -1526,6 +1542,46 @@ DEF_OP(VFRecp) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFRecpPrecision) {
|
||||
const auto Op = IROp->C<IR::IROp_VFRecpPrecision>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT((OpSize == IR::OpSize::i64Bit || OpSize == IR::OpSize::i32Bit) && ElementSize == IR::OpSize::i32Bit,
|
||||
"Unexpected sizes for operation.", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = OpSize == ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (IsScalar) {
|
||||
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
|
||||
// Not enough precision so we need to improve it with frecps
|
||||
frecpe(SubRegSize.Scalar, VTMP1.S(), Vector.S());
|
||||
frecps(SubRegSize.Scalar, VTMP2.S(), VTMP1.S(), Vector.S());
|
||||
fmul(SubRegSize.Scalar, Dst.S(), VTMP1.S(), VTMP2.S());
|
||||
return;
|
||||
}
|
||||
|
||||
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0f);
|
||||
// Element size is known to be 32bits
|
||||
fdiv(Dst.S(), VTMP1.S(), Vector.S());
|
||||
} else { // Vector operation - Opsize 64bits, elementsize 32bits
|
||||
if (HostSupportsRPRES) {
|
||||
frecpe(SubRegSize.Vector, VTMP1.D(), Vector.D());
|
||||
frecps(SubRegSize.Vector, VTMP2.D(), VTMP1.D(), Vector.D());
|
||||
fmul(SubRegSize.Vector, Dst.D(), VTMP1.D(), VTMP2.D());
|
||||
return;
|
||||
}
|
||||
|
||||
// No RPRES, so normal division
|
||||
fmov(SubRegSize.Vector, VTMP1.Q(), 1.0f);
|
||||
fdiv(SubRegSize.Vector, Dst.Q(), VTMP1.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFRSqrt) {
|
||||
const auto Op = IROp->C<IR::IROp_VFRSqrt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1595,6 +1651,50 @@ DEF_OP(VFRSqrt) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFRSqrtPrecision) {
|
||||
const auto Op = IROp->C<IR::IROp_VFRSqrtPrecision>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
|
||||
LOGMAN_THROW_A_FMT((OpSize == IR::OpSize::i64Bit || OpSize == IR::OpSize::i32Bit) && ElementSize == IR::OpSize::i32Bit,
|
||||
"Unexpected sizes for operation.", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (IsScalar) {
|
||||
if (HostSupportsRPRES) {
|
||||
frsqrte(SubRegSize.Scalar, VTMP1.S(), Vector.S());
|
||||
// Improve initial estimate which is not good enough.
|
||||
fmul(SubRegSize.Scalar, VTMP2.S(), VTMP1.S(), VTMP1.S());
|
||||
frsqrts(SubRegSize.Scalar, VTMP2.S(), VTMP2.S(), Vector.S());
|
||||
fmul(SubRegSize.Scalar, Dst.S(), VTMP1.S(), VTMP2.S());
|
||||
return;
|
||||
}
|
||||
|
||||
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0);
|
||||
// element size is known to be 32bits
|
||||
fsqrt(VTMP2.S(), Vector.S());
|
||||
fdiv(Dst.S(), VTMP1.S(), VTMP2.S());
|
||||
} else {
|
||||
if (HostSupportsRPRES) {
|
||||
frsqrte(SubRegSize.Vector, VTMP1.D(), Vector.D());
|
||||
// Improve initial estimate which is not good enough.
|
||||
fmul(SubRegSize.Vector, VTMP2.D(), VTMP1.D(), VTMP1.D());
|
||||
frsqrts(SubRegSize.Vector, VTMP2.D(), VTMP2.D(), Vector.D());
|
||||
fmul(SubRegSize.Vector, Dst.D(), VTMP1.D(), VTMP2.D());
|
||||
return;
|
||||
}
|
||||
fmov(SubRegSize.Vector, VTMP1.Q(), 1.0);
|
||||
fsqrt(SubRegSize.Vector, VTMP2.Q(), Vector.Q());
|
||||
fdiv(SubRegSize.Vector, Dst.Q(), VTMP1.Q(), VTMP2.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VNot) {
|
||||
const auto Op = IROp->C<IR::IROp_VNot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -4453,5 +4553,29 @@ DEF_OP(VFNMLS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFCopySign) {
|
||||
auto Op = IROp->C<IR::IROp_VFCopySign>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
ARMEmitter::VRegister Magnitude = GetVReg(Op->Vector1.ID());
|
||||
ARMEmitter::VRegister Sign = GetVReg(Op->Vector2.ID());
|
||||
|
||||
// We don't assign explicity to Dst but Dst and Magniture are tied to the same register.
|
||||
// Similar in semantics to C's copysignf.
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i64Bit:
|
||||
movi(SubRegSize, VTMP1.D(), 0x80, 24);
|
||||
bit(Magnitude.D(), Sign.D(), VTMP1.D());
|
||||
break;
|
||||
case IR::OpSize::i128Bit:
|
||||
movi(SubRegSize, VTMP1.Q(), 0x80, 24);
|
||||
bit(Magnitude.Q(), Sign.Q(), VTMP1.Q());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported element size for operation {}", __func__); FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1589,7 +1589,7 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
|
||||
const uint32_t Size = GetSrcBitSize(Op);
|
||||
const auto OpSize = Size == 64 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t UnmaskedConst;
|
||||
uint64_t UnmaskedConst {};
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op. But it's
|
||||
// equivalent to mask to the actual size of the op, that way we can bound
|
||||
@@ -2658,7 +2658,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
|
||||
Ref Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Result;
|
||||
Ref Result {};
|
||||
|
||||
if (Size != OpSize::i64Bit) {
|
||||
Src1 = _Bfe(OpSize::i64Bit, SizeBits, 0, Src1);
|
||||
@@ -3001,6 +3001,22 @@ void OpDispatchBuilder::SGDTOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, _Constant(GDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
|
||||
auto DestAddress = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
|
||||
// See SGDTOp, matches Linux in reported values
|
||||
uint64_t IDTAddress = 0xFFFFFE0000000000ULL;
|
||||
auto IDTStoreSize = OpSize::i64Bit;
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
// Mask off upper bits if 32-bit result.
|
||||
IDTAddress &= ~0U;
|
||||
IDTStoreSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
_StoreMemAutoTSO(GPRClass, OpSize::i16Bit, DestAddress, _Constant(0xfff));
|
||||
_StoreMemAutoTSO(GPRClass, IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, _Constant(IDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SMSWOp(OpcodeArgs) {
|
||||
const bool IsMemDst = DestIsMem(Op);
|
||||
|
||||
@@ -4166,21 +4182,97 @@ Ref OpDispatchBuilder::LoadEffectiveAddress(AddressMode A, bool AddSegmentBase,
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [this, &A]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(A, true),
|
||||
.Index = InvalidNode,
|
||||
};
|
||||
};
|
||||
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
// In the future this also needs to account for LRCPC3.
|
||||
bool SupportsRegIndex = Vector || !AtomicTSO;
|
||||
|
||||
// Try a constant offset. For 64-bit, this maps directly. For 32-bit, this
|
||||
// works only for displacements with magnitude < 16KB, since those bottom
|
||||
// addresses are reserved and therefore wrap around is invalid.
|
||||
// Loadstore rules:
|
||||
// Non-TSO GPR:
|
||||
// * LDR/STR: [Reg]
|
||||
// * LDR/STR: [Reg + Reg, {Shift <AccessSize>}]
|
||||
// * Can't use with 32-bit
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * Imm must be smaller than 16k with 32-bit
|
||||
// * LDUR/STUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// TODO: Also handle GPR TSO if we can guarantee the constant inlines.
|
||||
if (SupportsRegIndex) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && A.AddrSize == OpSize::i32Bit && GPRSize == OpSize::i32Bit;
|
||||
// TSO GPR:
|
||||
// * ARMv8.0:
|
||||
// LDAR/STLR: [Reg]
|
||||
// * FEAT_LRCPC:
|
||||
// LDAPR: [Reg]
|
||||
// * FEAT_LRCPC2:
|
||||
// LDAPUR/STLUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// Non-TSO Vector:
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * LDUR/STUR: [Reg + [-256,255]]
|
||||
//
|
||||
// TSO Vector:
|
||||
// * ARMv8.0:
|
||||
// Just DMB + previous
|
||||
// * FEAT_LRCPC3 (Unsupported by FEXCore currently):
|
||||
// LDAPUR/STLUR: [Reg + [-256,255]]
|
||||
|
||||
if ((A.AddrSize == OpSize::i64Bit) || Const_16K) {
|
||||
const auto AccessSizeAsImm = OpSizeToSize(AccessSize);
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [this](AddressMode A) -> AddressMode {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(B, true /* AddSegmentBase */, false),
|
||||
.Index = _InlineConstant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
|
||||
auto ScaledRegisterLoadstore = [this, &GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = _Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) & !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
}
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && GPRSizeMatchesAddrSize && Is32Bit;
|
||||
|
||||
if (!Is32Bit || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
@@ -4193,25 +4285,10 @@ AddressMode OpDispatchBuilder::SelectAddressMode(AddressMode A, bool AtomicTSO,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Try a (possibly scaled) register index.
|
||||
if (A.AddrSize == OpSize::i64Bit && A.Base && (A.Index || A.Segment) && !A.Offset &&
|
||||
(A.IndexScale == 1 || A.IndexScale == IR::OpSizeToSize(AccessSize))) {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = _Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(A, true),
|
||||
.Index = InvalidNode,
|
||||
};
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
@@ -4607,6 +4684,12 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LSLOp(OpcodeArgs) {
|
||||
// Emulate by always returning failure, this deviates from both Linux and Windows but
|
||||
// shouldn't be depended on by anything.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
IR::BreakDefinition Reason;
|
||||
bool SetRIPToNext = false;
|
||||
|
||||
@@ -302,6 +302,7 @@ public:
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void LSLOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
void SyscallOp(OpcodeArgs, bool IsSyscallInst);
|
||||
void ThunkOp(OpcodeArgs);
|
||||
@@ -417,6 +418,7 @@ public:
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
void SIDTOp(OpcodeArgs);
|
||||
void SMSWOp(OpcodeArgs);
|
||||
|
||||
enum class VectorOpType {
|
||||
@@ -434,6 +436,7 @@ public:
|
||||
|
||||
void VectorALUROp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void RSqrt3DNowOp(OpcodeArgs, bool Duplicate);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
|
||||
@@ -901,6 +904,15 @@ public:
|
||||
return Pair;
|
||||
}
|
||||
|
||||
Ref SHADataShuffle(Ref Src) {
|
||||
// SHA data shuffle matches PSHUFD shuffle where elements are inverted.
|
||||
// Because this shuffle mask gets reused multiple times per instruction, it's always a win to load the mask once and reuse it.
|
||||
const uint32_t Shuffle = 0b00'01'10'11;
|
||||
auto LookupIndexes =
|
||||
LoadAndCacheIndexedNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD, Shuffle * 16);
|
||||
return _VTBL1(OpSize::i128Bit, Src, LookupIndexes);
|
||||
}
|
||||
|
||||
RefPair AVX128_LoadSource_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
bool NeedsHigh, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
|
||||
@@ -783,7 +783,7 @@ void OpDispatchBuilder::AVX128_VZERO(OpcodeArgs) {
|
||||
if (IsVZEROALL) {
|
||||
// NOTE: Despite the name being VZEROALL, this will still only ever
|
||||
// zero out up to the first 16 registers (even on AVX-512, where we have 32 registers)
|
||||
Ref ZeroVector;
|
||||
Ref ZeroVector {};
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
// Explicitly not caching named vector zero. This ensures that every register gets movi #0.0 directly.
|
||||
|
||||
@@ -62,29 +62,36 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
auto Src2 = SHADataShuffle(Src);
|
||||
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
// The result is swizzled differently than expected
|
||||
Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
} else {
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
}
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
@@ -92,16 +99,16 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1c?
|
||||
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p with different key
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1m
|
||||
return Self.BitwiseAtLeastTwo(B, C, D);
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
|
||||
@@ -119,60 +126,92 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
f3,
|
||||
};
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
Ref Result {};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Ref ConstantVector {};
|
||||
switch (Imm8) {
|
||||
case 0:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
|
||||
break;
|
||||
case 1:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
|
||||
break;
|
||||
case 2:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
|
||||
break;
|
||||
case 3:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
|
||||
break;
|
||||
}
|
||||
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Ref Src1 = SHADataShuffle(Dest);
|
||||
Ref Src2 = SHADataShuffle(Src);
|
||||
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
|
||||
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
switch (Imm8) {
|
||||
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
|
||||
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
|
||||
case 1:
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
} else {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
|
||||
StoreResult(FPRClass, Op, Dest0, OpSize::iInvalid);
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -222,19 +261,28 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
|
||||
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
Result = _VSha256U1(Src1, Src2);
|
||||
} else {
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
|
||||
StoreResult(FPRClass, Op, D0, OpSize::iInvalid);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
|
||||
@@ -9,16 +9,16 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, true>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
|
||||
@@ -21,6 +21,11 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
@@ -36,6 +41,11 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x03, 1, &OpDispatchBuilder::LSLOp},
|
||||
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x0B, 1, &OpDispatchBuilder::INTOp},
|
||||
|
||||
@@ -626,17 +626,36 @@ void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i64Bit>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::RSqrt3DNowOp(OpcodeArgs, bool Duplicate) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto ElementSize = OpSize::i32Bit;
|
||||
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags);
|
||||
|
||||
// For the sqrt reciprocal in 3DNow!, if the source is negative,
|
||||
// then the result has the same sign as the source but the result is always calculated
|
||||
// as if the source was positive.
|
||||
Ref AbsSrc = _VFAbs(Size, ElementSize, Src);
|
||||
Ref PosRSqrt = _VFRSqrtPrecision(Size, ElementSize, AbsSrc);
|
||||
Ref Result = _VFCopySign(Size, ElementSize, PosRSqrt, Src);
|
||||
|
||||
if (Duplicate) {
|
||||
Result = _VDupElement(Size, ElementSize, Result, 0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
|
||||
// In the event of a scalar operation and a vector source, then
|
||||
// we can specify the entire vector length in order to avoid
|
||||
// unnecessary sign extension on the element to be operated on.
|
||||
// In the event of a memory operand, we load the exact element size.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(SrcSize, ElementSize, Src));
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags);
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(Size, ElementSize, Src));
|
||||
StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
@@ -676,8 +695,8 @@ void OpDispatchBuilder::VectorUnaryDuplicateOp(OpcodeArgs) {
|
||||
VectorUnaryDuplicateOpImpl(Op, IROp, ElementSize);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>(OpcodeArgs);
|
||||
// TODO: there's only one instantiation of this template. Lets remove it.
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVQOp(OpcodeArgs, VectorOpType VectorType) {
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
@@ -967,13 +986,17 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
// Special case element duplicate and broadcast to low or high 64-bits.
|
||||
return _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src, Shuffle & 0b11);
|
||||
}
|
||||
|
||||
case 0b00'00'10'10: {
|
||||
// Weird reverse low elements and broadcast to each half of the register
|
||||
Ref Tmp = _VUnZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
return _VZip(OpSize::i128Bit, OpSize::i32Bit, Tmp, Tmp);
|
||||
}
|
||||
case 0b00'00'11'10: {
|
||||
// First element duplicated and shifted in to the top.
|
||||
auto Dup = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dup, Src, 2);
|
||||
}
|
||||
case 0b00'01'00'01: {
|
||||
///< Weird reversed low elements and broadcast
|
||||
Ref Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
@@ -984,6 +1007,11 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
Ref Tmp = _VZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Tmp, Tmp, 4);
|
||||
}
|
||||
case 0b00'01'10'11: {
|
||||
// Inverse elements
|
||||
Ref Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i32Bit, Tmp, Tmp, 2);
|
||||
}
|
||||
case 0b00'10'00'10: {
|
||||
///< Weird reversed even elements and broadcast
|
||||
Ref Tmp = _VUnZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
@@ -1102,6 +1130,10 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
Ref Tmp = _VZip2(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Tmp, Tmp, 8);
|
||||
}
|
||||
case 0b10'11'00'01: {
|
||||
// Reverse each 64-bit lane.
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
}
|
||||
case 0b10'11'10'11: {
|
||||
///< Weird top two elements reverse and broadcast
|
||||
Ref Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src, Src);
|
||||
|
||||
@@ -40,7 +40,7 @@ Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW;
|
||||
Ref NewAbridgedFTW {};
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
Ref RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW);
|
||||
|
||||
@@ -67,41 +67,41 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -23,7 +23,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0x01, 1, X86InstInfo{"", TYPE_GROUP_7, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
// These two load segment register data
|
||||
{0x02, 1, X86InstInfo{"LAR", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
@@ -338,7 +338,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
const auto& FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
|
||||
// The lambda is converted to std::function. This is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
// NOTE: unique_ptr must be passed as a raw pointer since std::function requires lambda captures to be copyable
|
||||
|
||||
@@ -1881,6 +1881,12 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFCopySign OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Returns a vector where each element has has the magniture of each corresponding element in vector1 and the sign of vector 2."],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
@@ -1967,9 +1973,25 @@
|
||||
},
|
||||
|
||||
"FPR = VFRecp OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Reciprocal value - matches the precision required by the x86 spec.",
|
||||
"It has a relative error of at most 1.5 * 2^-12"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFRecpPrecision OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Similar to VFRecp but carrying more precision for 3DNow!",
|
||||
"It provides at least 14 bits precision, with a relative error of at most 2^-14"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i64Bit || RegisterSize == FEXCore::IR::OpSize::i32Bit",
|
||||
"ElementSize == FEXCore::IR::OpSize::i32Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VFSqrt OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1977,9 +1999,26 @@
|
||||
},
|
||||
|
||||
"FPR = VFRSqrt OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Reciprocal Square Root - matches the precision required by the x86 spec.",
|
||||
"It has a relative error of at most 1.5 * 2^-12"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFRSqrtPrecision OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Similar to VFRSqrt but carrying more precision for 3DNow!",
|
||||
"It provides at least 15 bits precision, with a relative error of at most 2^-15"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i64Bit || RegisterSize == FEXCore::IR::OpSize::i32Bit",
|
||||
"ElementSize == FEXCore::IR::OpSize::i32Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VCMPEQZ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
@@ -2656,8 +2695,33 @@
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i32Bit"
|
||||
},
|
||||
"FPR = VSha1C FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1C instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1M FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1M instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1P FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1P instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1SU1 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U0 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256U1 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U1 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit"
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, OpSize:$SrcSize": {
|
||||
|
||||
@@ -667,6 +667,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
@@ -682,6 +683,13 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMPAIR: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemPair>();
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value1_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value2_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
@@ -692,6 +700,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ private:
|
||||
};
|
||||
|
||||
IRDumper::IRDumper() {
|
||||
const auto DumpIRStr = DumpIR();
|
||||
const auto& DumpIRStr = DumpIR();
|
||||
if (DumpIRStr == "stderr" || DumpIRStr == "stdout" || DumpIRStr == "no") {
|
||||
// Intentionally do nothing
|
||||
} else if (DumpIRStr == "server") {
|
||||
|
||||
@@ -250,7 +250,7 @@ private:
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
RegisterClassType Class;
|
||||
uint8_t Reg;
|
||||
uint8_t Reg {};
|
||||
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
|
||||
@@ -42,7 +42,8 @@ constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
constexpr uint32_t LDSTREGISTER_MASK = 0b0011'1011'0010'0000'0000'1100'0000'0000;
|
||||
// Load/store register (register offset) (Rm encoded as xzr)
|
||||
constexpr uint32_t LDSTREGISTER_MASK = 0b0011'1111'1111'1111'1111'1100'0000'0000;
|
||||
constexpr uint32_t LDR_INST = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
constexpr uint32_t STR_INST = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
|
||||
|
||||
@@ -1,13 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#ifndef _WIN32
|
||||
#include <linux/magic.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/vfs.h>
|
||||
#include <time.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -15,10 +12,11 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#define BACKEND_OFF 0
|
||||
#define BACKEND_GPUVIS 1
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
#include <array>
|
||||
#include <limits.h>
|
||||
#include <time.h>
|
||||
#ifndef _WIN32
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
@@ -49,7 +47,6 @@ static inline uint64_t GetTime() {
|
||||
|
||||
#endif
|
||||
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
namespace FEXCore::Profiler {
|
||||
ProfilerBlock::ProfilerBlock(std::string_view const Format)
|
||||
: DurationBegin {GetTime()}
|
||||
@@ -114,35 +111,122 @@ void TraceObject(std::string_view const Format) {
|
||||
}
|
||||
}
|
||||
} // namespace GPUVis
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
#include "tracy/Tracy.hpp"
|
||||
namespace Tracy {
|
||||
static int EnableAfterFork = 0;
|
||||
static bool Enable = false;
|
||||
|
||||
void Init(std::string_view ProgramName, std::string_view ProgramPath) {
|
||||
const char* ProfileTargetName = getenv("FEX_PROFILE_TARGET_NAME"); // Match by application name
|
||||
const char* ProfileTargetPath = getenv("FEX_PROFILE_TARGET_PATH"); // Match by path suffix
|
||||
const char* WaitForFork = getenv("FEX_PROFILE_WAIT_FOR_FORK"); // Don't enable profiling until the process forks N times
|
||||
bool Matched = (ProfileTargetName && ProgramName == ProfileTargetName) || (ProfileTargetPath && ProgramPath.ends_with(ProfileTargetPath));
|
||||
if (Matched && WaitForFork) {
|
||||
EnableAfterFork = std::atoi(WaitForFork);
|
||||
}
|
||||
Enable = Matched && !EnableAfterFork;
|
||||
if (Enable) {
|
||||
tracy::StartupProfiler();
|
||||
LogMan::Msg::IFmt("Tracy profiling started");
|
||||
} else if (EnableAfterFork) {
|
||||
LogMan::Msg::IFmt("Tracy profiling will start after fork");
|
||||
}
|
||||
}
|
||||
|
||||
void PostForkAction(bool IsChild) {
|
||||
if (Enable) {
|
||||
// Tracy does not support multiprocess profiling
|
||||
LogMan::Msg::EFmt("Warning: Profiling a process with forks is not supported. Set the environment variable "
|
||||
"FEX_PROFILE_WAIT_FOR_FORK=<n> to start profiling after the n-th fork.");
|
||||
}
|
||||
|
||||
if (IsChild) {
|
||||
Enable = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if (EnableAfterFork > 1) {
|
||||
--EnableAfterFork;
|
||||
LogMan::Msg::IFmt("Tracy profiling will start after {} forks", EnableAfterFork);
|
||||
} else if (EnableAfterFork == 1) {
|
||||
Enable = true;
|
||||
EnableAfterFork = 0;
|
||||
tracy::StartupProfiler();
|
||||
LogMan::Msg::IFmt("Tracy profiling started");
|
||||
}
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
if (Tracy::Enable) {
|
||||
LogMan::Msg::IFmt("Stopping Tracy profiling");
|
||||
tracy::ShutdownProfiler();
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
if (Tracy::Enable) {
|
||||
TracyMessage(Format.data(), Format.size());
|
||||
}
|
||||
}
|
||||
} // namespace Tracy
|
||||
#else
|
||||
#error Unknown profiler backend
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
void Init() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
void Init(std::string_view ProgramName, std::string_view ProgramPath) {
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::Init();
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::Init(ProgramName, ProgramPath);
|
||||
#endif
|
||||
}
|
||||
|
||||
void PostForkAction(bool IsChild) {
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::PostForkAction(IsChild);
|
||||
#endif
|
||||
}
|
||||
|
||||
bool IsActive() {
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
// Always active
|
||||
return true;
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
// Active if previously enabled
|
||||
return Tracy::Enable;
|
||||
#endif
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::Shutdown();
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::Shutdown();
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format, Duration);
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::TraceObject(Format, Duration);
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format);
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::TraceObject(Format);
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::Profiler
|
||||
@@ -45,7 +45,7 @@ void Initialize() {
|
||||
return;
|
||||
}
|
||||
|
||||
auto DataDirectory = Config::GetTelemetryDirectory();
|
||||
const auto& DataDirectory = Config::GetTelemetryDirectory();
|
||||
|
||||
// Ensure the folder structure is created for our configuration
|
||||
if (!FHU::Filesystem::Exists(DataDirectory) && !FHU::Filesystem::CreateDirectories(DataDirectory)) {
|
||||
|
||||
@@ -36,6 +36,10 @@ class OpDispatchBuilder;
|
||||
class PassManager;
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
struct ThreadStats;
|
||||
};
|
||||
|
||||
namespace FEXCore::Core {
|
||||
|
||||
// Special-purpose replacement for std::unique_ptr to allow InternalThreadState to be standard layout.
|
||||
@@ -95,6 +99,9 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
|
||||
std::shared_mutex ObjectCacheRefCounter {};
|
||||
|
||||
// This pointer is owned by the frontend.
|
||||
FEXCore::Profiler::ThreadStats* ThreadStats {};
|
||||
|
||||
///< Data pointer for exclusive use by the frontend
|
||||
void* FrontendPtr;
|
||||
|
||||
|
||||
@@ -80,6 +80,10 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_CVTMAX_I32,
|
||||
NAMED_VECTOR_CVTMAX_I64,
|
||||
NAMED_VECTOR_F80_SIGN_MASK,
|
||||
NAMED_VECTOR_SHA1RNDS_K0,
|
||||
NAMED_VECTOR_SHA1RNDS_K1,
|
||||
NAMED_VECTOR_SHA1RNDS_K2,
|
||||
NAMED_VECTOR_SHA1RNDS_K3,
|
||||
|
||||
NAMED_VECTOR_CONST_POOL_MAX,
|
||||
// Beginning of named constants that don't have a constant pool backing.
|
||||
|
||||
@@ -1,18 +1,99 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <x86intrin.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#define FEXCORE_PROFILER_BACKEND_OFF 0
|
||||
#define FEXCORE_PROFILER_BACKEND_GPUVIS 1
|
||||
#define FEXCORE_PROFILER_BACKEND_TRACY 2
|
||||
|
||||
#if defined(ENABLE_FEXCORE_PROFILER) && FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
#include "tracy/Tracy.hpp"
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
// FEXCore live-stats
|
||||
constexpr uint8_t STATS_VERSION = 2;
|
||||
enum class AppType : uint8_t {
|
||||
LINUX_32,
|
||||
LINUX_64,
|
||||
WIN_ARM64EC,
|
||||
WIN_WOW64,
|
||||
};
|
||||
|
||||
struct ThreadStatsHeader {
|
||||
uint8_t Version;
|
||||
AppType app_type;
|
||||
uint8_t _pad[2];
|
||||
char fex_version[48];
|
||||
std::atomic<uint32_t> Head;
|
||||
std::atomic<uint32_t> Size;
|
||||
uint32_t Pad;
|
||||
};
|
||||
|
||||
struct ThreadStats {
|
||||
std::atomic<uint32_t> Next;
|
||||
std::atomic<uint32_t> TID;
|
||||
|
||||
// Accumulated time (In unscaled CPU cycles!)
|
||||
uint64_t AccumulatedJITTime;
|
||||
uint64_t AccumulatedSignalTime;
|
||||
|
||||
// Accumulated event counts
|
||||
uint64_t AccumulatedSIGBUSCount;
|
||||
uint64_t AccumulatedSMCCount;
|
||||
uint64_t AccumulatedFloatFallbackCount;
|
||||
};
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init();
|
||||
#ifdef _M_ARM_64
|
||||
/**
|
||||
* @brief Get the raw cycle counter with synchronizing isb.
|
||||
*
|
||||
* `CNTVCTSS_EL0` also does the same thing, but requires the FEAT_ECV feature.
|
||||
*/
|
||||
static inline uint64_t GetCycleCounter() {
|
||||
uint64_t Result {};
|
||||
__asm volatile(R"(
|
||||
isb;
|
||||
mrs %[Res], CNTVCT_EL0;
|
||||
)"
|
||||
: [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
#else
|
||||
static inline uint64_t GetCycleCounter() {
|
||||
unsigned dummy;
|
||||
uint64_t tsc = __rdtscp(&dummy);
|
||||
return tsc;
|
||||
}
|
||||
#endif
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init(std::string_view ProgramName, std::string_view ProgramPath);
|
||||
FEX_DEFAULT_VISIBILITY void PostForkAction(bool IsChild);
|
||||
FEX_DEFAULT_VISIBILITY bool IsActive();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
|
||||
|
||||
#define UniqueScopeName2(name, line) name##line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) ZoneNamedN(___tracy_scoped_zone, name, ::FEXCore::Profiler::IsActive())
|
||||
#else
|
||||
// A class that follows scoping rules to generate a profile duration block
|
||||
class ProfilerBlock final {
|
||||
public:
|
||||
@@ -25,18 +106,45 @@ private:
|
||||
std::string_view const Format;
|
||||
};
|
||||
|
||||
#define UniqueScopeName2(name, line) name##line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) FEXCore::Profiler::ProfilerBlock UniqueScopeName(ScopedBlock_, __LINE__)(name)
|
||||
#endif
|
||||
|
||||
template<typename T, size_t FlatOffset = 0>
|
||||
class AccumulationBlock final {
|
||||
public:
|
||||
AccumulationBlock(T* Stat)
|
||||
: Begin {GetCycleCounter()}
|
||||
, Stat {Stat} {}
|
||||
|
||||
~AccumulationBlock() {
|
||||
const auto Duration = GetCycleCounter() - Begin + FlatOffset;
|
||||
if (Stat) {
|
||||
auto ref = std::atomic_ref<T>(*Stat);
|
||||
ref.fetch_add(Duration, std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
uint64_t Begin;
|
||||
T* Stat;
|
||||
};
|
||||
|
||||
#define FEXCORE_PROFILE_ACCUMULATION(ThreadState, Stat) \
|
||||
FEXCore::Profiler::AccumulationBlock<decltype(ThreadState->ThreadStats->Stat)> UniqueScopeName(ScopedAccumulation_, __LINE__)( \
|
||||
ThreadState->ThreadStats ? &ThreadState->ThreadStats->Stat : nullptr);
|
||||
#define FEXCORE_PROFILE_INSTANT_INCREMENT(ThreadState, Stat, value) \
|
||||
do { \
|
||||
if (ThreadState->ThreadStats) { \
|
||||
ThreadState->ThreadStats->Stat += value; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#else
|
||||
[[maybe_unused]]
|
||||
static void Init() {}
|
||||
static void Init(std::string_view ProgramName, std::string_view ProgramPath) {}
|
||||
[[maybe_unused]]
|
||||
static void PostForkAction(bool IsChild) {}
|
||||
[[maybe_unused]]
|
||||
static void Shutdown() {}
|
||||
[[maybe_unused]]
|
||||
@@ -50,5 +158,12 @@ static void TraceObject(std::string_view const, uint64_t) {}
|
||||
#define FEXCORE_PROFILE_SCOPED(...) \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_PROFILE_ACCUMULATION(...) \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_PROFILE_INSTANT_INCREMENT(...) \
|
||||
do { \
|
||||
} while (0)
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::Profiler
|
||||
@@ -0,0 +1,106 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
|
||||
#include <functional>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
namespace fextl {
|
||||
|
||||
/**
|
||||
* Equivalent to std::move_only_function but uses FEXCore::Allocator routines
|
||||
* for non-function pointers.
|
||||
*/
|
||||
template<typename F, void* (*Alloc)(size_t, size_t) = ::FEXCore::Allocator::aligned_alloc, void (*Dealloc)(void*) = ::FEXCore::Allocator::aligned_free>
|
||||
class move_only_function;
|
||||
|
||||
template<typename R, typename... Args, void* (*Alloc)(size_t, size_t), void (*Dealloc)(void*)>
|
||||
class move_only_function<R(Args...), Alloc, Dealloc> {
|
||||
public:
|
||||
template<typename F>
|
||||
requires std::is_invocable_r_v<R, F, Args...>
|
||||
move_only_function(F&& f) noexcept(std::is_nothrow_move_constructible_v<F>) {
|
||||
if constexpr (std::is_convertible_v<F, R (*)(Args...)>) {
|
||||
// Argument is a function pointer, a captureless lambda, or a stateless function object.
|
||||
// std::function can store these without allocation
|
||||
internal = std::move(f);
|
||||
} else if constexpr (std::is_nothrow_constructible_v<std::function<R(Args...)>, F>) {
|
||||
// If construction is guaranteed not to throw an exception, this implies
|
||||
// the std::function implementation won't allocate memory!
|
||||
internal = std::move(f);
|
||||
} else {
|
||||
// Other arguments require allocation, which is a problem since
|
||||
// std::function doesn't allow allocator customization. Implementations
|
||||
// are generally able to avoid allocation for lambdas with a single
|
||||
// pointer capture however. We can exploit this special case by wrapping
|
||||
// the actual argument in a lambda that points an external storage
|
||||
// location.
|
||||
|
||||
static_assert(!std::is_pointer_v<F>, "Pointer types must manually be dereferenced");
|
||||
|
||||
// First, relocate argument to a location returned from FEX's allocators
|
||||
using Fnoref = std::remove_reference_t<F>;
|
||||
storage = Alloc(std::alignment_of_v<Fnoref>, sizeof(Fnoref));
|
||||
auto moved_lambda = new (storage) Fnoref {std::move(f)};
|
||||
|
||||
// Second, wrap the relocated argument in a single-capture lambda
|
||||
auto wrapped_lambda = [moved_lambda](Args... args) {
|
||||
return (*moved_lambda)(std::forward<Args>(args)...);
|
||||
};
|
||||
|
||||
// Third, assign the result to std::function, ensuring it's indeed
|
||||
// allocation-free by checking for nothrow-constructibility
|
||||
static_assert(noexcept(internal = std::move(wrapped_lambda)), "This implementation of std::function "
|
||||
"does not support implementing "
|
||||
"fextl::move_only_function");
|
||||
internal = std::move(wrapped_lambda);
|
||||
|
||||
// Finally, if a destructor must be called, generate a pointer to its destructor
|
||||
if constexpr (!std::is_trivially_destructible_v<Fnoref>) {
|
||||
internal_destructor = [](move_only_function* self) {
|
||||
reinterpret_cast<Fnoref*>(self->storage)->~Fnoref();
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
move_only_function() noexcept {}
|
||||
move_only_function(std::nullptr_t) noexcept {}
|
||||
move_only_function(const move_only_function&) = delete;
|
||||
move_only_function(move_only_function&& other) noexcept {
|
||||
*this = std::move(other);
|
||||
}
|
||||
|
||||
move_only_function& operator=(move_only_function&& other) noexcept {
|
||||
if (!other && internal_destructor) {
|
||||
this->~move_only_function();
|
||||
}
|
||||
internal = std::exchange(other.internal, nullptr);
|
||||
internal_destructor = std::exchange(other.internal_destructor, nullptr);
|
||||
storage = std::exchange(other.storage, nullptr);
|
||||
return *this;
|
||||
}
|
||||
|
||||
~move_only_function() {
|
||||
if (internal_destructor) {
|
||||
internal_destructor(this);
|
||||
}
|
||||
Dealloc(storage);
|
||||
}
|
||||
|
||||
R operator()(Args... args) const {
|
||||
return internal(std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
explicit operator bool() const noexcept {
|
||||
return (bool)internal;
|
||||
}
|
||||
|
||||
private:
|
||||
std::function<R(Args...)> internal;
|
||||
void (*internal_destructor)(move_only_function*) = nullptr;
|
||||
void* storage = nullptr;
|
||||
};
|
||||
} // namespace fextl
|
||||
@@ -913,10 +913,10 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Advanced SIMD load/store
|
||||
TEST_SINGLE(ld3<SubRegSize::i32Bit>(VReg::v26, VReg::v27, VReg::v28, 0, Reg::r30, 12), "ld3 {v26.s, v27.s, v28.s}[0], [x30], #12");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(VReg::v26, VReg::v27, VReg::v28, 0, Reg::r30, 24), "ld3 {v26.d, v27.d, v28.d}[0], [x30], #24");
|
||||
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(VReg::v26, VReg::v27, VReg::v28, 15, Reg::r30, 1), "ld3 {v26.b, v27.b, v28.b}[15], [x30], #3");
|
||||
TEST_SINGLE(ld3<SubRegSize::i16Bit>(VReg::v26, VReg::v27, VReg::v28, 7, Reg::r30, 2), "ld3 {v26.h, v27.h, v28.h}[7], [x30], #6");
|
||||
TEST_SINGLE(ld3<SubRegSize::i32Bit>(VReg::v26, VReg::v27, VReg::v28, 3, Reg::r30, 4), "ld3 {v26.s, v27.s, v28.s}[3], [x30], #12");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(VReg::v26, VReg::v27, VReg::v28, 1, Reg::r30, 8), "ld3 {v26.d, v27.d, v28.d}[1], [x30], #24");
|
||||
TEST_SINGLE(ld3<SubRegSize::i8Bit>(VReg::v26, VReg::v27, VReg::v28, 15, Reg::r30, 3), "ld3 {v26.b, v27.b, v28.b}[15], [x30], #3");
|
||||
TEST_SINGLE(ld3<SubRegSize::i16Bit>(VReg::v26, VReg::v27, VReg::v28, 7, Reg::r30, 6), "ld3 {v26.h, v27.h, v28.h}[7], [x30], #6");
|
||||
TEST_SINGLE(ld3<SubRegSize::i32Bit>(VReg::v26, VReg::v27, VReg::v28, 3, Reg::r30, 12), "ld3 {v26.s, v27.s, v28.s}[3], [x30], #12");
|
||||
TEST_SINGLE(ld3<SubRegSize::i64Bit>(VReg::v26, VReg::v27, VReg::v28, 1, Reg::r30, 24), "ld3 {v26.d, v27.d, v28.d}[1], [x30], #24");
|
||||
|
||||
TEST_SINGLE(ld3r<SubRegSize::i8Bit>(DReg::d31, DReg::d0, DReg::d1, Reg::r30, 3), "ld3r {v31.8b, v0.8b, v1.8b}, [x30], #3");
|
||||
TEST_SINGLE(ld3r<SubRegSize::i8Bit>(DReg::d26, DReg::d27, DReg::d28, Reg::r30, 3), "ld3r {v26.8b, v27.8b, v28.8b}, [x30], #3");
|
||||
|
||||
@@ -2963,12 +2963,12 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE2 integer multiply long") {
|
||||
|
||||
// TEST_SINGLE(pmullb(SubRegSize::i8Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullb z30.b, z29.b, z28.b");
|
||||
TEST_SINGLE(pmullb(SubRegSize::i16Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullb z30.h, z29.b, z28.b");
|
||||
TEST_SINGLE(pmullb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullb z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(pmullb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullb z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(pmullb(SubRegSize::i64Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullb z30.d, z29.s, z28.s");
|
||||
|
||||
// TEST_SINGLE(pmullt(SubRegSize::i8Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullt z30.b, z29.b, z28.b");
|
||||
TEST_SINGLE(pmullt(SubRegSize::i16Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullt z30.h, z29.b, z28.b");
|
||||
TEST_SINGLE(pmullt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullt z30.s, z29.h, z28.h");
|
||||
// TEST_SINGLE(pmullt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullt z30.s, z29.h, z28.h");
|
||||
TEST_SINGLE(pmullt(SubRegSize::i64Bit, ZReg::z30, ZReg::z29, ZReg::z28), "pmullt z30.d, z29.s, z28.s");
|
||||
|
||||
// TEST_SINGLE(smullb(SubRegSize::i8Bit, ZReg::z30, ZReg::z29, ZReg::z28), "smullb z30.b, z29.b, z28.b");
|
||||
|
||||
@@ -110,15 +110,14 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: Barriers") {
|
||||
|
||||
TEST_SINGLE(isb(), "isb");
|
||||
|
||||
// vixl has a decoding bug claiming these are system level instructions.
|
||||
TEST_SINGLE(sb(), "sb (System)");
|
||||
TEST_SINGLE(tcommit(), "tcommit (System)");
|
||||
TEST_SINGLE(sb(), "sb");
|
||||
TEST_SINGLE(tcommit(), "tcommit");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System register move") {
|
||||
// vixl doesn't have decoding for a bunch of these.
|
||||
// Also most of these aren't writeable from el0, just testing the encoding.
|
||||
TEST_SINGLE(msr(SystemRegister::CTR_EL0, Reg::r30), "msr S3_3_c0_c0_1, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::DCZID_EL0, Reg::r30), "msr S3_3_c0_c0_7, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::DCZID_EL0, Reg::r30), "msr dczid_el0, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::TPIDR_EL0, Reg::r30), "msr S3_3_c13_c0_2, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::RNDR, Reg::r30), "msr rndr, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::RNDRRS, Reg::r30), "msr rndrrs, x30");
|
||||
@@ -129,7 +128,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System register move") {
|
||||
TEST_SINGLE(msr(SystemRegister::CNTVCT_EL0, Reg::r30), "msr S3_3_c14_c0_2, x30");
|
||||
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::CTR_EL0), "mrs x30, S3_3_c0_c0_1");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::DCZID_EL0), "mrs x30, S3_3_c0_c0_7");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::DCZID_EL0), "mrs x30, dczid_el0");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::TPIDR_EL0), "mrs x30, S3_3_c13_c0_2");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::RNDR), "mrs x30, rndr");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::RNDRRS), "mrs x30, rndrrs");
|
||||
|
||||
@@ -125,6 +125,16 @@ inline fextl::string GetFilename(const fextl::string& Path) {
|
||||
return Path.substr(LastSeparator + 1);
|
||||
}
|
||||
|
||||
inline std::string_view GetFilename(std::string_view Path) {
|
||||
auto LastSeparator = Path.rfind('/');
|
||||
if (LastSeparator == fextl::string::npos) {
|
||||
// No separator. Likely relative `.`, `..`, `<Application Name>`, or empty string.
|
||||
return Path;
|
||||
}
|
||||
|
||||
return Path.substr(LastSeparator + 1);
|
||||
}
|
||||
|
||||
inline fextl::string ParentPath(const fextl::string& Path) {
|
||||
auto LastSeparator = Path.rfind('/');
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEX::ArgLoader {
|
||||
void FEX::ArgLoader::ArgLoader::Load() {
|
||||
void FEX::ArgLoader::ArgLoader::PreLoad() {
|
||||
RemainingArgs.clear();
|
||||
ProgramArguments.clear();
|
||||
if (Type == LoadType::WITHOUT_FEXLOADER_PARSER) {
|
||||
|
||||
@@ -18,10 +18,13 @@ public:
|
||||
, Type {Type}
|
||||
, argc {argc}
|
||||
, argv {argv} {
|
||||
Load();
|
||||
PreLoad();
|
||||
}
|
||||
|
||||
void Load() override;
|
||||
void Load() override {
|
||||
// Intentional no-op.
|
||||
}
|
||||
void PreLoad();
|
||||
void LoadWithoutArguments();
|
||||
fextl::vector<fextl::string> Get() {
|
||||
return RemainingArgs;
|
||||
|
||||
@@ -0,0 +1,435 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/**
|
||||
* Helper framework to enable asynchronous IO operations on file descriptor objects (networking, files).
|
||||
*
|
||||
* Strongly inspired by Boost.Asio.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cassert>
|
||||
#include <chrono>
|
||||
#include <optional>
|
||||
#include <poll.h>
|
||||
#include <span>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
namespace fasio {
|
||||
|
||||
enum class error {
|
||||
success,
|
||||
timeout, // User-specified timeout expired
|
||||
eof, // Permanently reached end of data stream (e.g. because socket connection was closed by peer)
|
||||
invalid, // Invalid input parameters
|
||||
generic_errno // Read errno for details
|
||||
};
|
||||
|
||||
/**
|
||||
* This selects which action to trigger when returning from a reactor callback.
|
||||
* The default (drop) will drop the callback so that the caller can register
|
||||
* a new one.
|
||||
*/
|
||||
enum class post_callback {
|
||||
drop, // Drop the callback
|
||||
repeat, // Continue using the same callback
|
||||
stop_reactor, // Triggers exit from run()
|
||||
};
|
||||
|
||||
/**
|
||||
* Core event loop for asynchronous code. Corresponds to asio::io_context,
|
||||
* specialized for multiplexing file descriptors via ppoll().
|
||||
*
|
||||
* A reactor tracks a set of file descriptors and calls user-provided callbacks
|
||||
* when they become ready. For example, the callback for a network socket will
|
||||
* be called when data is ready to be reveived on the socket.
|
||||
*
|
||||
*/
|
||||
struct poll_reactor {
|
||||
private:
|
||||
std::vector<pollfd> PollFDs;
|
||||
std::optional<int> CurrentFD; // FD that is currently being processed
|
||||
|
||||
int AsyncStopRequest[2] = {-1, -1};
|
||||
|
||||
// Maps FD to callback
|
||||
fextl::map<int, fextl::move_only_function<post_callback(error)>> callbacks;
|
||||
|
||||
struct Event {
|
||||
pollfd FD;
|
||||
bool Erase = false;
|
||||
bool Insert = false;
|
||||
};
|
||||
std::vector<Event> QueuedEvents;
|
||||
|
||||
public:
|
||||
~poll_reactor() {
|
||||
if (AsyncStopRequest[0]) {
|
||||
::close(AsyncStopRequest[0]);
|
||||
::close(AsyncStopRequest[1]);
|
||||
}
|
||||
}
|
||||
|
||||
// Adds an internal FD to wake up and exit the reactor when stop_async() is called from any thread.
|
||||
void enable_async_stop() {
|
||||
::pipe(AsyncStopRequest);
|
||||
PollFDs.push_back(pollfd {.fd = AsyncStopRequest[0], .events = POLLHUP, .revents = 0});
|
||||
callbacks[AsyncStopRequest[0]] = [](error) {
|
||||
return post_callback::stop_reactor;
|
||||
};
|
||||
}
|
||||
|
||||
void stop_async() {
|
||||
if (AsyncStopRequest[1] == -1) {
|
||||
ERROR_AND_DIE_FMT("Tried to use stop_async without calling enable_async_stop during setup");
|
||||
}
|
||||
// Wake up run() thread by closing this pipe endpoint
|
||||
::close(AsyncStopRequest[1]);
|
||||
}
|
||||
|
||||
error run(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
// Process events queued before entering wait loop
|
||||
update_fd_list();
|
||||
|
||||
timespec ts = to_timespec(Timeout.value_or(std::chrono::nanoseconds {0}));
|
||||
|
||||
while (true) {
|
||||
int Result = ::ppoll(PollFDs.data(), PollFDs.size(), Timeout ? &ts : nullptr, nullptr);
|
||||
|
||||
if (Result < 0) {
|
||||
if (errno == EINTR || errno == EAGAIN) {
|
||||
continue;
|
||||
}
|
||||
callbacks.clear();
|
||||
return error::generic_errno;
|
||||
} else if (Result == 0) {
|
||||
callbacks.clear();
|
||||
return error::timeout;
|
||||
} else {
|
||||
bool exit_requested = false;
|
||||
|
||||
// Walk the FDs and see if we got any results
|
||||
for (auto& ActiveFD : PollFDs) {
|
||||
if (ActiveFD.revents == 0) {
|
||||
continue;
|
||||
}
|
||||
if (Result-- == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (ActiveFD.revents & POLLIN) {
|
||||
// NOTE: For sockets, this is triggered on close, too. Pipes only report POLLHUP, however.
|
||||
CurrentFD = ActiveFD.fd;
|
||||
|
||||
auto Callback = std::move(callbacks[ActiveFD.fd]);
|
||||
if (!Callback) {
|
||||
ERROR_AND_DIE_FMT("Data available for reading on FD {} but no read callback registered", ActiveFD.fd);
|
||||
}
|
||||
auto Ret = Callback(error::success);
|
||||
if (Ret == post_callback::repeat) {
|
||||
callbacks[ActiveFD.fd] = std::move(Callback);
|
||||
} else if (Ret == post_callback::drop) {
|
||||
// If no new callback was registered, drop the FD from the list and skip any remaining events
|
||||
if (!callbacks.contains(ActiveFD.fd)) {
|
||||
QueuedEvents.push_back(Event {.FD = {.fd = ActiveFD.fd}, .Erase = true});
|
||||
ActiveFD.revents = 0;
|
||||
}
|
||||
} else if (Ret == post_callback::stop_reactor) {
|
||||
exit_requested = true;
|
||||
}
|
||||
CurrentFD.reset();
|
||||
}
|
||||
if (ActiveFD.revents & (POLLHUP | POLLERR | POLLNVAL | POLLRDHUP)) {
|
||||
auto Callback = std::move(callbacks[ActiveFD.fd]);
|
||||
if (Callback) {
|
||||
exit_requested |= (Callback(error::eof) == post_callback::stop_reactor);
|
||||
}
|
||||
// Error or hangup, erase the socket from our list
|
||||
QueuedEvents.push_back(Event {.FD = {.fd = ActiveFD.fd}, .Erase = true});
|
||||
}
|
||||
|
||||
ActiveFD.revents = 0;
|
||||
}
|
||||
|
||||
if (exit_requested) {
|
||||
callbacks.clear();
|
||||
return error::success;
|
||||
}
|
||||
|
||||
update_fd_list();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void bind_handler(pollfd FD, fextl::move_only_function<post_callback(error)> Callback) {
|
||||
[[maybe_unused]] auto Previous = std::exchange(callbacks[FD.fd], std::move(Callback));
|
||||
assert(!Previous && "May not queue multiple async operations");
|
||||
|
||||
// Add the FD to the poll list if it's not already contained
|
||||
if (CurrentFD != FD.fd && PollFDs.end() == std::find_if(PollFDs.begin(), PollFDs.end(), [&](auto& Prev) { return FD.fd == Prev.fd; })) {
|
||||
QueuedEvents.push_back(Event {.FD = FD, .Insert = true});
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
timespec to_timespec(std::chrono::nanoseconds Duration) {
|
||||
timespec Timespec {};
|
||||
auto Seconds = std::chrono::duration_cast<std::chrono::seconds>(Duration);
|
||||
Timespec.tv_sec = Seconds.count();
|
||||
Timespec.tv_nsec = std::chrono::duration_cast<std::chrono::nanoseconds>(Duration - Seconds).count();
|
||||
return Timespec;
|
||||
}
|
||||
|
||||
void update_fd_list() {
|
||||
for (auto& Event : QueuedEvents) {
|
||||
if (Event.Erase) {
|
||||
std::iter_swap(std::find_if(PollFDs.begin(), PollFDs.end(), [&](auto& FD) { return FD.fd == Event.FD.fd; }), std::prev(PollFDs.end()));
|
||||
PollFDs.pop_back();
|
||||
callbacks.erase(Event.FD.fd);
|
||||
}
|
||||
|
||||
if (Event.Insert) {
|
||||
PollFDs.push_back(Event.FD);
|
||||
}
|
||||
}
|
||||
QueuedEvents.clear();
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Corresponds to asio::mutable_buffer.
|
||||
*/
|
||||
struct mutable_buffer {
|
||||
std::span<std::byte> Data;
|
||||
mutable_buffer* Next = nullptr;
|
||||
|
||||
// Optional FD to send/receive via ancillary buffer.
|
||||
// This may only be used with non-empty data, and there may only be up to one FD per buffer chain
|
||||
std::optional<int*> FD;
|
||||
|
||||
size_t size() const {
|
||||
size_t Ret = 0;
|
||||
const mutable_buffer* Current = this;
|
||||
do {
|
||||
Ret += Current->Data.size_bytes();
|
||||
Current = Current->Next;
|
||||
} while (Current);
|
||||
|
||||
if (Ret == 0) {
|
||||
assert(!FD);
|
||||
}
|
||||
return Ret;
|
||||
}
|
||||
|
||||
int consume_fd() {
|
||||
assert(FD);
|
||||
return **std::exchange(FD, std::nullopt);
|
||||
}
|
||||
|
||||
mutable_buffer& operator+=(size_t NumBytes) {
|
||||
mutable_buffer* Current = this;
|
||||
while (Current->Next && NumBytes >= Current->Data.size_bytes()) {
|
||||
NumBytes -= Data.size_bytes();
|
||||
Current = Current->Next;
|
||||
assert(Current->FD == std::nullopt);
|
||||
}
|
||||
auto FD = std::exchange(this->FD, std::nullopt);
|
||||
*this = *Current;
|
||||
Data = Data.subspan(std::min(Data.size_bytes(), NumBytes));
|
||||
this->FD = FD;
|
||||
return *Current;
|
||||
}
|
||||
|
||||
size_t count_chunks() const {
|
||||
size_t Ret = 1;
|
||||
const mutable_buffer* Current = this;
|
||||
while (Current->Next) {
|
||||
Current = Current->Next;
|
||||
++Ret;
|
||||
}
|
||||
return Ret;
|
||||
}
|
||||
};
|
||||
|
||||
inline mutable_buffer Chained(std::span<mutable_buffer> Buffers) {
|
||||
for (size_t i = 0; i + 1 < Buffers.size(); ++i) {
|
||||
Buffers[i].Next = &Buffers[i + 1];
|
||||
}
|
||||
return Buffers[0];
|
||||
}
|
||||
|
||||
/**
|
||||
* Corresponds to asio::dynamic_vector_buffer.
|
||||
*/
|
||||
struct dynamic_vector_buffer {
|
||||
fextl::vector<std::byte>& Data;
|
||||
|
||||
// Maximum number of bytes to grow to
|
||||
size_t max_size = Data.capacity();
|
||||
};
|
||||
|
||||
/**
|
||||
* Asynchronously reads data from the given stream until MatchPredicate reports a match. The read
|
||||
* is queued to the stream's reactor and will progress whenever data is available.
|
||||
*
|
||||
* MatchPredicate must have the signature pair<Iter, bool>(Iter, Iter):
|
||||
* - The input iterators provide the range of new data bytes
|
||||
* - The returned boolean indicates if a match was found
|
||||
* - The returned iterator is the match location or the location at which to continue testing after the next read
|
||||
*
|
||||
* The read data will be appended to Buffers. Data past the match returned from the last read data will also be included.
|
||||
*
|
||||
* Corresponds to asio::async_read_until.
|
||||
*/
|
||||
template<typename AsyncReadStream, typename MatchPredicate, typename OnComplete>
|
||||
requires std::is_invocable_r_v<void, OnComplete, error, size_t>
|
||||
void async_read_until(AsyncReadStream& Stream, dynamic_vector_buffer Buffers, MatchPredicate Predicate, OnComplete UserCallback) {
|
||||
struct Callback {
|
||||
size_t BeginPos;
|
||||
size_t EndPos;
|
||||
AsyncReadStream& Stream;
|
||||
dynamic_vector_buffer Buffers;
|
||||
MatchPredicate Predicate;
|
||||
OnComplete UserCallback;
|
||||
|
||||
void operator()(error Err, size_t BytesRead, std::optional<int> FD) {
|
||||
if (Err != error::success) {
|
||||
UserCallback(Err, 0);
|
||||
return;
|
||||
}
|
||||
|
||||
// Start with the predicate check to avoid fetching data unnecessarily
|
||||
EndPos += BytesRead;
|
||||
if (EndPos != BeginPos) {
|
||||
auto Begin = Buffers.Data.begin() + BeginPos;
|
||||
auto End = Buffers.Data.begin() + EndPos;
|
||||
auto [It, Found] = Predicate(Begin, End);
|
||||
BeginPos = It - Buffers.Data.begin();
|
||||
if (Found) {
|
||||
Buffers.Data.resize(EndPos); // Shrink down to size of data actually received
|
||||
UserCallback(error::success, BeginPos);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// Fill the entire remaining capacity, or resize for a minimum of 512 bytes
|
||||
auto BytesToRead = std::max<size_t>(std::min(Buffers.Data.capacity(), Buffers.max_size) - EndPos, 512);
|
||||
if (Buffers.Data.size() + BytesToRead > Buffers.max_size) {
|
||||
ERROR_AND_DIE_FMT("Out of buffer space");
|
||||
}
|
||||
|
||||
Buffers.Data.resize(EndPos + BytesToRead);
|
||||
|
||||
// Queue data read.
|
||||
// On completion, Reader will check if enough data was received and will queue more reads if needed.
|
||||
Stream.async_read_some(mutable_buffer {std::span {Buffers.Data}}, *this);
|
||||
}
|
||||
};
|
||||
|
||||
// Check existing data for a predicate match, then initiate async reading if necessary
|
||||
Callback {0, Buffers.Data.size(), Stream, Buffers, std::move(Predicate), std::move(UserCallback)}(error::success, 0, std::nullopt);
|
||||
}
|
||||
|
||||
using read_callback = fextl::move_only_function<void(error, size_t, std::optional<int>)>;
|
||||
|
||||
/**
|
||||
* Synchronously reads fixed-length data from the given Stream.
|
||||
*
|
||||
* The length is inferred from the size of the output buffer(s).
|
||||
*
|
||||
* Corresponds to asio::read.
|
||||
*/
|
||||
template<typename AsyncReadStream>
|
||||
std::size_t read(AsyncReadStream& Stream, mutable_buffer Buffers, error& ec) {
|
||||
size_t TotalBytesRead = 0;
|
||||
while (Buffers.size() != 0 || Buffers.FD) {
|
||||
auto BytesRead = Stream.read_some(Buffers, ec);
|
||||
TotalBytesRead += BytesRead;
|
||||
if (Buffers.FD) {
|
||||
(void)Buffers.consume_fd();
|
||||
}
|
||||
Buffers += BytesRead;
|
||||
if (ec != error::success) {
|
||||
return TotalBytesRead;
|
||||
}
|
||||
}
|
||||
ec = error::success;
|
||||
return TotalBytesRead;
|
||||
}
|
||||
|
||||
/**
|
||||
* Synchronously writes fixed-length data to the given Stream.
|
||||
*
|
||||
* The length is inferred from the size of the input buffer(s).
|
||||
*
|
||||
* Corresponds to asio::write.
|
||||
*/
|
||||
template<typename AsyncReadStream>
|
||||
std::size_t write(AsyncReadStream& Stream, mutable_buffer Buffers, error& ec) {
|
||||
size_t TotalBytesWritten = 0;
|
||||
while (Buffers.size() != 0 || Buffers.FD) {
|
||||
auto BytesWritten = Stream.write_some(Buffers, ec);
|
||||
TotalBytesWritten += BytesWritten;
|
||||
if (Buffers.FD) {
|
||||
(void)Buffers.consume_fd();
|
||||
}
|
||||
Buffers += BytesWritten;
|
||||
if (ec != error::success) {
|
||||
return TotalBytesWritten;
|
||||
}
|
||||
}
|
||||
ec = error::success;
|
||||
return TotalBytesWritten;
|
||||
}
|
||||
|
||||
/**
|
||||
* Owning RAII wrapper around a file descriptor.
|
||||
*
|
||||
* Corresponds to asio::posix::descriptor.
|
||||
*/
|
||||
struct posix_descriptor {
|
||||
poll_reactor* Reactor = nullptr;
|
||||
int FD = -1;
|
||||
|
||||
posix_descriptor(poll_reactor& Reactor, int FD)
|
||||
: Reactor(&Reactor)
|
||||
, FD(FD) {}
|
||||
|
||||
posix_descriptor(posix_descriptor&& Other)
|
||||
: Reactor(Other.Reactor)
|
||||
, FD(std::exchange(Other.FD, -1)) {}
|
||||
|
||||
posix_descriptor& operator=(posix_descriptor&& Other) {
|
||||
posix_descriptor::~posix_descriptor();
|
||||
Reactor = Other.Reactor;
|
||||
FD = std::exchange(Other.FD, -1);
|
||||
return *this;
|
||||
}
|
||||
|
||||
~posix_descriptor() {
|
||||
if (FD != -1) {
|
||||
::close(FD);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wait until there is data available to read on this object, then execute the given callback
|
||||
*/
|
||||
template<typename Fn>
|
||||
requires std::is_invocable_r_v<post_callback, Fn, error>
|
||||
void async_wait(Fn Callback) {
|
||||
Reactor->bind_handler(
|
||||
pollfd {
|
||||
.fd = FD,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
},
|
||||
std::move(Callback));
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace fasio
|
||||
@@ -0,0 +1,286 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/**
|
||||
* Socket wrappers for asynchronous programming with fasio
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <Common/Async.h>
|
||||
|
||||
#include <sys/socket.h>
|
||||
#include <sys/un.h>
|
||||
|
||||
namespace fasio {
|
||||
|
||||
/**
|
||||
* Non-owning wrapper around a socket.
|
||||
*
|
||||
* Corresponds to asio::local::stream_protocol::socket.
|
||||
*/
|
||||
struct tcp_socket {
|
||||
poll_reactor* Reactor = nullptr;
|
||||
int FD;
|
||||
|
||||
// Constructor for synchronous and asynchronous operation
|
||||
tcp_socket(poll_reactor& Reactor_, int FD_)
|
||||
: Reactor(&Reactor_)
|
||||
, FD(FD_) {}
|
||||
|
||||
// Constructor for purely synchronous operation
|
||||
tcp_socket(int FD_)
|
||||
: FD(FD_) {}
|
||||
|
||||
/**
|
||||
* Queues an asynchronous operation that will run the completion callback
|
||||
* once at least one byte of data was received
|
||||
*/
|
||||
template<typename OnComplete>
|
||||
requires std::is_invocable_r_v<void, OnComplete, error, size_t, std::optional<int>>
|
||||
void async_read_some(mutable_buffer Buffers, OnComplete UserCallback) {
|
||||
auto Callback = [Buffers, Socket = FD, UserCallback = std::move(UserCallback)](error ec) mutable {
|
||||
if (ec != error::success) {
|
||||
UserCallback(ec, 0, std::nullopt);
|
||||
return post_callback::drop;
|
||||
}
|
||||
|
||||
auto BytesRead = read_some_from_fd(Buffers, ec, Socket);
|
||||
if (ec != error::success) {
|
||||
UserCallback(ec, BytesRead, std::nullopt);
|
||||
} else {
|
||||
UserCallback(ec, BytesRead, Buffers.FD ? std::optional {**Buffers.FD} : std::nullopt);
|
||||
}
|
||||
return post_callback::drop;
|
||||
};
|
||||
|
||||
Reactor->bind_handler(
|
||||
pollfd {
|
||||
.fd = FD,
|
||||
.events = POLLIN | POLLPRI | POLLRDHUP,
|
||||
.revents = 0,
|
||||
},
|
||||
std::move(Callback));
|
||||
}
|
||||
|
||||
/**
|
||||
* Blocks until at least one byte of data was received
|
||||
*/
|
||||
size_t read_some(const mutable_buffer& Buffers, error& ec) {
|
||||
return read_some_from_fd(Buffers, ec, FD);
|
||||
}
|
||||
|
||||
/**
|
||||
* Blocks until at least one byte of data was sent
|
||||
*/
|
||||
size_t write_some(const mutable_buffer& Buffers, error& ec) {
|
||||
auto iov = (iovec*)alloca(sizeof(mutable_buffer) * Buffers.count_chunks());
|
||||
size_t NumIovs = 0;
|
||||
for (auto Buffer = &Buffers; Buffer; Buffer = Buffer->Next) {
|
||||
iov[NumIovs].iov_base = Buffer->Data.data();
|
||||
iov[NumIovs].iov_len = Buffer->Data.size_bytes();
|
||||
++NumIovs;
|
||||
}
|
||||
msghdr msg {
|
||||
.msg_name = nullptr,
|
||||
.msg_namelen = 0,
|
||||
.msg_iov = iov,
|
||||
.msg_iovlen = NumIovs,
|
||||
};
|
||||
|
||||
// Setup the ancillary buffer. This is where we will be getting pipe FDs
|
||||
// We only need 4 bytes for the FD
|
||||
constexpr size_t CMSG_SIZE = CMSG_SPACE(sizeof(int));
|
||||
union AncillaryBuffer {
|
||||
cmsghdr Header;
|
||||
uint8_t Buffer[CMSG_SIZE];
|
||||
};
|
||||
AncillaryBuffer AncBuf {};
|
||||
|
||||
if (Buffers.FD) {
|
||||
// Enable ancillary buffer
|
||||
msg.msg_control = AncBuf.Buffer;
|
||||
msg.msg_controllen = CMSG_SIZE;
|
||||
|
||||
// Now we need to setup the ancillary buffer data. We are only sending an FD
|
||||
cmsghdr* cmsg = CMSG_FIRSTHDR(&msg);
|
||||
cmsg->cmsg_len = CMSG_LEN(sizeof(int));
|
||||
cmsg->cmsg_level = SOL_SOCKET;
|
||||
cmsg->cmsg_type = SCM_RIGHTS;
|
||||
|
||||
// We are giving the daemon the write side of the pipe
|
||||
memcpy(CMSG_DATA(cmsg), Buffers.FD.value(), sizeof(int));
|
||||
}
|
||||
|
||||
ssize_t Ret;
|
||||
do {
|
||||
Ret = ::sendmsg(FD, &msg, 0);
|
||||
} while (Ret < 0 && (errno == EINTR || errno == EAGAIN));
|
||||
if (Ret < 0) {
|
||||
ec = error::generic_errno;
|
||||
return 0;
|
||||
}
|
||||
ec = error::success;
|
||||
return Ret;
|
||||
}
|
||||
|
||||
private:
|
||||
static size_t read_some_from_fd(const mutable_buffer& Buffers, error& ec, int FD) {
|
||||
auto iov = (iovec*)alloca(sizeof(mutable_buffer) * Buffers.count_chunks());
|
||||
size_t NumIovs = 0;
|
||||
for (auto Buffer = &Buffers; Buffer; Buffer = Buffer->Next) {
|
||||
iov[NumIovs].iov_base = Buffer->Data.data();
|
||||
iov[NumIovs].iov_len = Buffer->Data.size_bytes();
|
||||
++NumIovs;
|
||||
}
|
||||
msghdr msg {
|
||||
.msg_name = nullptr,
|
||||
.msg_namelen = 0,
|
||||
.msg_iov = iov,
|
||||
.msg_iovlen = NumIovs,
|
||||
};
|
||||
|
||||
// If requested, set up a 4-byte ancillary buffer for receiving a file descriptor
|
||||
constexpr size_t CMSG_SIZE = CMSG_SPACE(sizeof(int));
|
||||
union AncillaryBuffer {
|
||||
cmsghdr Header;
|
||||
uint8_t Buffer[CMSG_SIZE];
|
||||
};
|
||||
AncillaryBuffer AncBuf {};
|
||||
|
||||
if (Buffers.FD) {
|
||||
// Enable ancillary buffer
|
||||
msg.msg_control = AncBuf.Buffer;
|
||||
msg.msg_controllen = CMSG_SIZE;
|
||||
}
|
||||
|
||||
ssize_t BytesRead;
|
||||
do {
|
||||
BytesRead = ::recvmsg(FD, &msg, 0);
|
||||
} while (BytesRead < 0 && (errno == EINTR || errno == EAGAIN));
|
||||
if (BytesRead < 0) {
|
||||
if (errno != 0) {
|
||||
ec = error::generic_errno;
|
||||
return 0;
|
||||
}
|
||||
} else if (BytesRead == 0) {
|
||||
ec = error::eof;
|
||||
return 0;
|
||||
}
|
||||
|
||||
struct cmsghdr* cmsg = CMSG_FIRSTHDR(&msg);
|
||||
if (Buffers.FD &&
|
||||
(cmsg == nullptr || cmsg->cmsg_len != CMSG_LEN(sizeof(int)) || cmsg->cmsg_level != SOL_SOCKET || cmsg->cmsg_type != SCM_RIGHTS)) {
|
||||
ec = error::invalid;
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (Buffers.FD) {
|
||||
memcpy(*Buffers.FD, CMSG_DATA(cmsg), sizeof(FD));
|
||||
}
|
||||
|
||||
ec = error::success;
|
||||
return BytesRead;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Owning wrapper around a server socket that listens for connections after
|
||||
* creation. Clients can be accepted asynchronously using async_accept().
|
||||
*
|
||||
* Corresponds to asio::local::stream_protocol::acceptor.
|
||||
*/
|
||||
struct tcp_acceptor {
|
||||
poll_reactor& Reactor;
|
||||
int FD;
|
||||
|
||||
tcp_acceptor(tcp_acceptor&& other)
|
||||
: Reactor(other.Reactor)
|
||||
, FD(other.FD) {
|
||||
other.FD = -1;
|
||||
}
|
||||
|
||||
~tcp_acceptor() {
|
||||
if (FD != -1) {
|
||||
::close(FD);
|
||||
}
|
||||
}
|
||||
|
||||
tcp_acceptor& operator=(tcp_acceptor&& other) {
|
||||
FD = std::exchange(other.FD, -1);
|
||||
return *this;
|
||||
}
|
||||
|
||||
static std::optional<tcp_acceptor> create(poll_reactor& Reactor, bool abstract, std::string_view Name, int MaxPending = SOMAXCONN) {
|
||||
// Create the initial unix socket
|
||||
int FD = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (FD == -1) {
|
||||
return {};
|
||||
}
|
||||
|
||||
sockaddr_un addr {};
|
||||
addr.sun_family = AF_UNIX;
|
||||
|
||||
if (Name.size() > sizeof(addr.sun_path) - 1) {
|
||||
ERROR_AND_DIE_FMT("Invalid FEXServer socket name: {}", Name);
|
||||
}
|
||||
|
||||
auto NameEnd = addr.sun_path;
|
||||
if (!abstract) {
|
||||
// sun_path is null-terminated
|
||||
NameEnd = std::copy(Name.begin(), Name.end(), addr.sun_path);
|
||||
*NameEnd++ = 0;
|
||||
} else {
|
||||
// Abstract AF_UNIX sockets start with \0 but aren't null-terminated
|
||||
addr.sun_path[0] = 0;
|
||||
NameEnd = std::copy(Name.begin(), Name.end(), addr.sun_path + 1);
|
||||
}
|
||||
|
||||
// Bind the socket to the path
|
||||
int Result = bind(FD, reinterpret_cast<sockaddr*>(&addr), sizeof(addr.sun_family) + (NameEnd - addr.sun_path));
|
||||
if (Result == -1) {
|
||||
::close(FD);
|
||||
return {};
|
||||
}
|
||||
|
||||
Result = ::listen(FD, MaxPending);
|
||||
if (Result == -1) {
|
||||
::close(FD);
|
||||
return {};
|
||||
}
|
||||
|
||||
return tcp_acceptor(Reactor, FD);
|
||||
}
|
||||
|
||||
void async_accept(fextl::move_only_function<post_callback(error, std::optional<tcp_socket>)> OnAccept) {
|
||||
Reactor.bind_handler(
|
||||
{
|
||||
.fd = FD,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
},
|
||||
[ServerFD = FD, &Reactor = Reactor, OnAccept = std::move(OnAccept)](error ec) mutable {
|
||||
if (ec != error::success) {
|
||||
return post_callback::drop;
|
||||
}
|
||||
|
||||
sockaddr_storage Addr {};
|
||||
socklen_t AddrSize {};
|
||||
int NewFD;
|
||||
do {
|
||||
NewFD = ::accept(ServerFD, reinterpret_cast<sockaddr*>(&Addr), &AddrSize);
|
||||
} while (NewFD < 0 && (errno == EINTR || errno == EAGAIN));
|
||||
if (NewFD < 0) {
|
||||
return OnAccept(error::generic_errno, std::nullopt);
|
||||
}
|
||||
|
||||
return OnAccept(error::success, tcp_socket {Reactor, NewFD});
|
||||
});
|
||||
}
|
||||
|
||||
private:
|
||||
tcp_acceptor(poll_reactor& Reactor_, int FD_)
|
||||
: Reactor(Reactor_)
|
||||
, FD(FD_) {}
|
||||
};
|
||||
static_assert(!std::is_copy_constructible_v<tcp_acceptor>);
|
||||
static_assert(!std::is_copy_assignable_v<tcp_acceptor>);
|
||||
|
||||
} // namespace fasio
|
||||
@@ -7,7 +7,8 @@ set(SRCS
|
||||
EnvironmentLoader.cpp
|
||||
HostFeatures.cpp
|
||||
JSONPool.cpp
|
||||
StringUtil.cpp)
|
||||
StringUtil.cpp
|
||||
Profiler.cpp)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND SRCS
|
||||
|
||||
+70
-11
@@ -66,11 +66,8 @@ static constexpr std::pair<std::string_view, FEXCore::Config::ConfigOption> Conf
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
|
||||
void SaveLayerToJSON(const fextl::string& Filename, FEXCore::Config::Layer* const Layer) {
|
||||
char Buffer[4096];
|
||||
char* Dest {};
|
||||
Dest = json_objOpen(Buffer, nullptr);
|
||||
Dest = json_objOpen(Dest, "Config");
|
||||
static char* SaveLayerToJSON(char* JsonBuffer, const FEXCore::Config::Layer* Layer) {
|
||||
JsonBuffer = json_objOpen(JsonBuffer, "Config");
|
||||
for (auto& it : Layer->GetOptionMap()) {
|
||||
std::string_view Name {};
|
||||
for (auto& name_it : ConfigLookup) {
|
||||
@@ -80,13 +77,30 @@ void SaveLayerToJSON(const fextl::string& Filename, FEXCore::Config::Layer* cons
|
||||
}
|
||||
}
|
||||
for (auto& var : it.second) {
|
||||
Dest = json_str(Dest, Name.data(), var.c_str());
|
||||
JsonBuffer = json_str(JsonBuffer, Name.data(), var.c_str());
|
||||
}
|
||||
}
|
||||
return json_objClose(JsonBuffer);
|
||||
}
|
||||
|
||||
void SaveLayerToJSON(const fextl::string& Filename, const FEXCore::Config::Layer* Layer, const fextl::unordered_map<fextl::string, bool>& HostLibs) {
|
||||
char Buffer[4096];
|
||||
char* Dest {};
|
||||
Dest = json_objOpen(Buffer, nullptr);
|
||||
|
||||
Dest = SaveLayerToJSON(Dest, Layer);
|
||||
|
||||
Dest = json_objOpen(Dest, "ThunksDB");
|
||||
for (auto& [Name, Enabled] : HostLibs) {
|
||||
Dest = json_int(Dest, Name.c_str(), Enabled);
|
||||
}
|
||||
Dest = json_objClose(Dest);
|
||||
|
||||
Dest = json_objClose(Dest);
|
||||
json_end(Dest);
|
||||
|
||||
LogMan::Throw::AFmt(Dest <= std::end(Buffer), "Exceeded JSON buffer size");
|
||||
|
||||
auto File = FEXCore::File::File(Filename.c_str(),
|
||||
FEXCore::File::FileModes::WRITE | FEXCore::File::FileModes::CREATE | FEXCore::File::FileModes::TRUNCATE);
|
||||
|
||||
@@ -95,6 +109,37 @@ void SaveLayerToJSON(const fextl::string& Filename, FEXCore::Config::Layer* cons
|
||||
}
|
||||
}
|
||||
|
||||
void SaveLayerToJSON(const fextl::string& Filename, const FEXCore::Config::Layer* Layer) {
|
||||
fextl::unordered_map<fextl::string, bool> HostLibsDB;
|
||||
|
||||
// Load existing ThunksDB entry to persist it
|
||||
{
|
||||
fextl::vector<char> FileData;
|
||||
if (!FEXCore::FileLoading::LoadFile(FileData, Filename)) {
|
||||
goto WriteConfig;
|
||||
}
|
||||
|
||||
// Find bounds of previously existing Config entry (if any)
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(FileData, Pool);
|
||||
if (!json) {
|
||||
goto WriteConfig;
|
||||
}
|
||||
|
||||
const json_t* ThunksDB = json_getProperty(json, "ThunksDB");
|
||||
if (!ThunksDB) {
|
||||
goto WriteConfig;
|
||||
}
|
||||
|
||||
for (const json_t* Item = json_getChild(ThunksDB); Item != nullptr; Item = json_getSibling(Item)) {
|
||||
HostLibsDB.emplace(json_getName(Item), (json_getInteger(Item) != 0));
|
||||
}
|
||||
}
|
||||
|
||||
WriteConfig:
|
||||
SaveLayerToJSON(Filename, Layer, HostLibsDB);
|
||||
}
|
||||
|
||||
// Application loaders
|
||||
class OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
@@ -108,7 +153,7 @@ class MainLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type, const char* ConfigFile);
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigFile);
|
||||
|
||||
void Load() override;
|
||||
|
||||
@@ -168,7 +213,7 @@ MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
, Config {std::move(ConfigFile)} {}
|
||||
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type, const char* ConfigFile)
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigFile)
|
||||
: OptionMapper(Type)
|
||||
, Config {ConfigFile} {}
|
||||
|
||||
@@ -259,7 +304,7 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* F
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(const char* AppConfig) {
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig) {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_USER_OVERRIDE, AppConfig);
|
||||
}
|
||||
|
||||
@@ -418,8 +463,15 @@ void LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgsLoader, fextl::
|
||||
}
|
||||
|
||||
const char* AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig && FHU::Filesystem::Exists(AppConfig)) {
|
||||
FEXCore::Config::AddLayer(CreateUserOverrideLayer(AppConfig));
|
||||
if (AppConfig) {
|
||||
fextl::string AppConfigStr = AppConfig;
|
||||
if (IsPortable && FHU::Filesystem::IsRelative(AppConfig)) {
|
||||
AppConfigStr = PortableInfo.InterpreterPath + AppConfigStr;
|
||||
}
|
||||
|
||||
if (FHU::Filesystem::Exists(AppConfigStr)) {
|
||||
FEXCore::Config::AddLayer(CreateUserOverrideLayer(AppConfigStr));
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(CreateEnvironmentLayer(envp));
|
||||
@@ -503,6 +555,13 @@ fextl::string GetConfigDirectory(bool Global, const PortableInformation& Portabl
|
||||
const char* ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (PortableInfo.IsPortable && (Global || !ConfigOverride)) {
|
||||
return fextl::fmt::format("{}/fex-emu/", PortableInfo.InterpreterPath);
|
||||
} else if (PortableInfo.IsPortable && ConfigOverride && !Global) {
|
||||
fextl::string AppConfigStr = ConfigOverride;
|
||||
if (PortableInfo.IsPortable && FHU::Filesystem::IsRelative(AppConfigStr)) {
|
||||
AppConfigStr = PortableInfo.InterpreterPath + AppConfigStr;
|
||||
}
|
||||
|
||||
return AppConfigStr;
|
||||
}
|
||||
|
||||
fextl::string ConfigDir;
|
||||
|
||||
@@ -19,7 +19,8 @@ public:
|
||||
protected:
|
||||
};
|
||||
|
||||
void SaveLayerToJSON(const fextl::string& Filename, FEXCore::Config::Layer* const Layer);
|
||||
void SaveLayerToJSON(const fextl::string& Filename, const FEXCore::Config::Layer* Layer);
|
||||
void SaveLayerToJSON(const fextl::string& Filename, const FEXCore::Config::Layer* Layer, const fextl::unordered_map<fextl::string, bool>& HostLibs);
|
||||
|
||||
struct ApplicationNames {
|
||||
// This is the full path to the program (if it exists).
|
||||
@@ -76,7 +77,7 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* File = nullptr);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(const char* AppConfig);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/AsyncNet.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FEXServerClient.h"
|
||||
|
||||
@@ -25,60 +26,31 @@
|
||||
|
||||
namespace FEXServerClient {
|
||||
int RequestPIDFDPacket(int ServerSocket, PacketType Type) {
|
||||
fasio::tcp_socket Socket {ServerSocket};
|
||||
FEXServerRequestPacket Req {
|
||||
.Header {
|
||||
.Type = Type,
|
||||
},
|
||||
};
|
||||
|
||||
int Result = write(ServerSocket, &Req, sizeof(Req.BasicRequest));
|
||||
if (Result != -1) {
|
||||
// Wait for success response with SCM_RIGHTS
|
||||
|
||||
FEXServerResultPacket Res {};
|
||||
struct iovec iov {
|
||||
.iov_base = &Res, .iov_len = sizeof(Res),
|
||||
};
|
||||
|
||||
struct msghdr msg {
|
||||
.msg_name = nullptr, .msg_namelen = 0, .msg_iov = &iov, .msg_iovlen = 1,
|
||||
};
|
||||
|
||||
// Setup the ancillary buffer. This is where we will be getting pipe FDs
|
||||
// We only need 4 bytes for the FD
|
||||
constexpr size_t CMSG_SIZE = CMSG_SPACE(sizeof(int));
|
||||
union AncillaryBuffer {
|
||||
struct cmsghdr Header;
|
||||
uint8_t Buffer[CMSG_SIZE];
|
||||
};
|
||||
AncillaryBuffer AncBuf {};
|
||||
|
||||
// Now link to our ancilllary buffer
|
||||
msg.msg_control = AncBuf.Buffer;
|
||||
msg.msg_controllen = CMSG_SIZE;
|
||||
|
||||
ssize_t DataResult = recvmsg(ServerSocket, &msg, 0);
|
||||
if (DataResult > 0) {
|
||||
// Now that we have the data, we can extract the FD from the ancillary buffer
|
||||
struct cmsghdr* cmsg = CMSG_FIRSTHDR(&msg);
|
||||
|
||||
// Do some error checking
|
||||
if (cmsg == nullptr || cmsg->cmsg_len != CMSG_LEN(sizeof(int)) || cmsg->cmsg_level != SOL_SOCKET || cmsg->cmsg_type != SCM_RIGHTS) {
|
||||
// Couldn't get a socket
|
||||
} else {
|
||||
// Check for Success.
|
||||
// If type error was returned then the FEXServer doesn't have a log to pipe in to
|
||||
if (Res.Header.Type == PacketType::TYPE_SUCCESS) {
|
||||
// Now that we know the cmsg is sane, read the FD
|
||||
int NewFD {};
|
||||
memcpy(&NewFD, CMSG_DATA(cmsg), sizeof(NewFD));
|
||||
return NewFD;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Send request
|
||||
fasio::error ec;
|
||||
write(Socket, fasio::mutable_buffer {std::as_writable_bytes(std::span {&Req, 1})}, ec);
|
||||
if (ec != fasio::error::success) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
return -1;
|
||||
// Wait for success response and log FD
|
||||
FEXServerResultPacket Res {};
|
||||
fasio::mutable_buffer ResBuffer {std::as_writable_bytes(std::span {&Res, 1})};
|
||||
int NewFD = -1;
|
||||
ResBuffer.FD = &NewFD;
|
||||
read(Socket, ResBuffer, ec);
|
||||
if (ec != fasio::error::success || Res.Header.Type != PacketType::TYPE_SUCCESS) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
return NewFD;
|
||||
}
|
||||
|
||||
static int ServerFD {-1};
|
||||
@@ -220,7 +192,7 @@ int ConnectToServer(ConnectionOption ConnectionOption) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
bool SetupClient(char* InterpreterPath) {
|
||||
bool SetupClient(std::string_view InterpreterPath) {
|
||||
ServerFD = FEXServerClient::ConnectToAndStartServer(InterpreterPath);
|
||||
if (ServerFD == -1) {
|
||||
return false;
|
||||
@@ -238,7 +210,7 @@ bool SetupClient(char* InterpreterPath) {
|
||||
return true;
|
||||
}
|
||||
|
||||
int ConnectToAndStartServer(char* InterpreterPath) {
|
||||
int ConnectToAndStartServer(std::string_view InterpreterPath) {
|
||||
int ServerFD = ConnectToServer(ConnectionOption::NoPrintConnectionError);
|
||||
if (ServerFD == -1) {
|
||||
// Couldn't connect to the server. Start one
|
||||
@@ -250,7 +222,7 @@ int ConnectToAndStartServer(char* InterpreterPath) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
fextl::string FEXServerPath = FHU::Filesystem::ParentPath(InterpreterPath) + "/FEXServer";
|
||||
fextl::string FEXServerPath = fextl::fmt::format("{}/FEXServer", InterpreterPath);
|
||||
// Check if a local FEXServer next to FEXInterpreter exists
|
||||
// If it does then it takes priority over the installed one
|
||||
if (!FHU::Filesystem::Exists(FEXServerPath)) {
|
||||
@@ -270,10 +242,13 @@ int ConnectToAndStartServer(char* InterpreterPath) {
|
||||
// Child
|
||||
close(fds[0]); // Close read end of pipe
|
||||
|
||||
const char* argv[2];
|
||||
const char* argv[4];
|
||||
|
||||
auto pipe_string = fextl::fmt::format("{}", fds[1]);
|
||||
argv[0] = FEXServerPath.c_str();
|
||||
argv[1] = nullptr;
|
||||
argv[1] = "--wait_pipe";
|
||||
argv[2] = pipe_string.c_str();
|
||||
argv[3] = nullptr;
|
||||
|
||||
if (execvp(argv[0], (char* const*)argv) == -1) {
|
||||
// Let the parent know that we couldn't execute for some reason
|
||||
|
||||
@@ -56,14 +56,14 @@ fextl::string GetServerSocketName();
|
||||
fextl::string GetServerSocketPath();
|
||||
int GetServerFD();
|
||||
|
||||
bool SetupClient(char* InterpreterPath);
|
||||
bool SetupClient(std::string_view InterpreterPath);
|
||||
|
||||
/**
|
||||
* @brief Connect to and start a FEXServer instance if required
|
||||
*
|
||||
* @return socket FD for communicating with server
|
||||
*/
|
||||
int ConnectToAndStartServer(char* InterpreterPath);
|
||||
int ConnectToAndStartServer(std::string_view InterpreterPath);
|
||||
|
||||
enum class ConnectionOption {
|
||||
Default,
|
||||
|
||||
@@ -543,10 +543,14 @@ FEXCore::HostFeatures FetchHostFeatures(FEX::CPUFeatures& Features, bool Support
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
HostFeatures.SupportsCLZERO = false;
|
||||
// Simulator doesn't support SHA
|
||||
HostFeatures.SupportsSHA = false;
|
||||
// simulator has a hardcoded ZVA size of 64-bytes.
|
||||
HostFeatures.SupportsCLZERO = true;
|
||||
HostFeatures.SupportsAES = true;
|
||||
HostFeatures.SupportsCRC = true;
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsSHA = true;
|
||||
HostFeatures.SupportsPMULL_128Bit = true;
|
||||
HostFeatures.SupportsAES256 = true;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/Profiler.h"
|
||||
#include "git_version.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
namespace FEX::Profiler {
|
||||
void StatAllocBase::SaveHeader(FEXCore::Profiler::AppType AppType) {
|
||||
if (!Base) {
|
||||
return;
|
||||
}
|
||||
|
||||
Head = reinterpret_cast<FEXCore::Profiler::ThreadStatsHeader*>(Base);
|
||||
Head->Size.store(CurrentSize, std::memory_order_relaxed);
|
||||
Head->Version = FEXCore::Profiler::STATS_VERSION;
|
||||
|
||||
std::string_view GitString = GIT_DESCRIBE_STRING;
|
||||
strncpy(Head->fex_version, GitString.data(), std::min(GitString.size(), sizeof(Head->fex_version)));
|
||||
Head->app_type = AppType;
|
||||
|
||||
Stats = reinterpret_cast<FEXCore::Profiler::ThreadStats*>(reinterpret_cast<uint64_t>(Base) + sizeof(FEXCore::Profiler::ThreadStatsHeader));
|
||||
|
||||
RemainingSlots = TotalSlotsFromSize();
|
||||
}
|
||||
|
||||
bool StatAllocBase::AllocateMoreSlots() {
|
||||
const auto OriginalSlotCount = TotalSlotsFromSize();
|
||||
|
||||
uint32_t NewSize = FrontendAllocateSlots(CurrentSize * 2);
|
||||
|
||||
if (NewSize == CurrentSize) {
|
||||
return false;
|
||||
}
|
||||
|
||||
CurrentSize = NewSize;
|
||||
Head->Size.store(CurrentSize, std::memory_order_relaxed);
|
||||
RemainingSlots = TotalSlotsFromSize() - OriginalSlotCount;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
FEXCore::Profiler::ThreadStats* StatAllocBase::AllocateSlot(uint32_t TID) {
|
||||
if (!RemainingSlots) {
|
||||
if (!AllocateMoreSlots()) {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
// Find a free slot
|
||||
store_memory_barrier();
|
||||
FEXCore::Profiler::ThreadStats* AllocatedSlot {};
|
||||
for (size_t i = 0; i < TotalSlotsFromSize(); ++i) {
|
||||
AllocatedSlot = &Stats[i];
|
||||
if (AllocatedSlot->TID.load(std::memory_order_relaxed) == 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
--RemainingSlots;
|
||||
|
||||
// Slot might be reused, just zero it now.
|
||||
memset(AllocatedSlot, 0, sizeof(*AllocatedSlot));
|
||||
|
||||
// TID != 0 means slot is allocated.
|
||||
AllocatedSlot->TID.store(TID, std::memory_order_relaxed);
|
||||
|
||||
// Setup singly-linked list
|
||||
if (Head->Head.load(std::memory_order_relaxed) == 0) {
|
||||
Head->Head.store(OffsetFromStat(AllocatedSlot), std::memory_order_relaxed);
|
||||
} else {
|
||||
StatTail->Next.store(OffsetFromStat(AllocatedSlot), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
// Update the tail.
|
||||
StatTail = AllocatedSlot;
|
||||
return AllocatedSlot;
|
||||
}
|
||||
|
||||
void StatAllocBase::DeallocateSlot(FEXCore::Profiler::ThreadStats* AllocatedSlot) {
|
||||
if (!AllocatedSlot) {
|
||||
return;
|
||||
}
|
||||
|
||||
// TID == 0 will signal the reader to ignore this slot & deallocate it!
|
||||
AllocatedSlot->TID.store(0, std::memory_order_relaxed);
|
||||
|
||||
store_memory_barrier();
|
||||
|
||||
const auto SlotOffset = OffsetFromStat(AllocatedSlot);
|
||||
const auto AllocatedSlotNext = AllocatedSlot->Next.load(std::memory_order_relaxed);
|
||||
|
||||
const bool IsTail = AllocatedSlot == StatTail;
|
||||
|
||||
// Update the linked list.
|
||||
if (Head->Head == SlotOffset) {
|
||||
Head->Head.store(AllocatedSlotNext, std::memory_order_relaxed);
|
||||
if (IsTail) {
|
||||
StatTail = nullptr;
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < TotalSlotsFromSize(); ++i) {
|
||||
auto Slot = &Stats[i];
|
||||
auto NextSlotOffset = Slot->Next.load(std::memory_order_relaxed);
|
||||
|
||||
if (NextSlotOffset == SlotOffset) {
|
||||
Slot->Next.store(AllocatedSlotNext, std::memory_order_relaxed);
|
||||
|
||||
if (IsTail) {
|
||||
// This slot is now the tail.
|
||||
StatTail = Slot;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
++RemainingSlots;
|
||||
}
|
||||
|
||||
} // namespace FEX::Profiler
|
||||
@@ -0,0 +1,69 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: Common|Profiler
|
||||
desc: Frontend profiler common code
|
||||
$end_info$
|
||||
*/
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static inline void store_memory_barrier() {
|
||||
asm volatile("dmb ishst;" ::: "memory");
|
||||
}
|
||||
|
||||
#else
|
||||
static inline void store_memory_barrier() {
|
||||
// Intentionally empty.
|
||||
// x86 is strongly memory ordered with regular loadstores. No need for barrier.
|
||||
}
|
||||
#endif
|
||||
|
||||
namespace FEX::Profiler {
|
||||
class StatAllocBase {
|
||||
protected:
|
||||
FEXCore::Profiler::ThreadStats* AllocateSlot(uint32_t TID);
|
||||
void DeallocateSlot(FEXCore::Profiler::ThreadStats* AllocatedSlot);
|
||||
|
||||
uint32_t OffsetFromStat(FEXCore::Profiler::ThreadStats* Stat) const {
|
||||
return reinterpret_cast<uint64_t>(Stat) - reinterpret_cast<uint64_t>(Base);
|
||||
}
|
||||
uint32_t TotalSlotsFromSize() const {
|
||||
return (CurrentSize - sizeof(FEXCore::Profiler::ThreadStatsHeader)) / sizeof(FEXCore::Profiler::ThreadStats) - 1;
|
||||
}
|
||||
static uint32_t TotalSlotsFromSize(uint32_t Size) {
|
||||
return (Size - sizeof(FEXCore::Profiler::ThreadStatsHeader)) / sizeof(FEXCore::Profiler::ThreadStats) - 1;
|
||||
}
|
||||
|
||||
static uint32_t SlotIndexFromOffset(uint32_t Offset) {
|
||||
return (Offset - sizeof(FEXCore::Profiler::ThreadStatsHeader)) / sizeof(FEXCore::Profiler::ThreadStats);
|
||||
}
|
||||
|
||||
void SaveHeader(FEXCore::Profiler::AppType AppType);
|
||||
|
||||
void* Base {};
|
||||
uint32_t CurrentSize {};
|
||||
FEXCore::Profiler::ThreadStatsHeader* Head {};
|
||||
FEXCore::Profiler::ThreadStats* Stats {};
|
||||
FEXCore::Profiler::ThreadStats* StatTail {};
|
||||
uint32_t RemainingSlots {};
|
||||
|
||||
// Limited to 4MB which should be a few hundred threads of tracking capability.
|
||||
// I (Sonicadvance1) wanted to reserve 128MB of VA space because it's cheap, but ran in to a bug when running WINE.
|
||||
// WINE allocates [0x7fff'fe00'0000, 0x7fff'ffff'0000) which /consistently/ overlaps with FEX's sigaltstack.
|
||||
// This only occurs when this stat allocation size is large as the top-down allocation pushes the alt-stack further.
|
||||
// Additionally, only occurs on 48-bit VA systems, as mmap on lesser VA will fail regardless.
|
||||
// TODO: Bump allocation size up once FEXCore's allocator can first use the 128TB of blocked VA space on 48-bit systems.
|
||||
constexpr static uint32_t MAX_STATS_SIZE = 4 * 1024 * 1024;
|
||||
|
||||
private:
|
||||
virtual uint32_t FrontendAllocateSlots(uint32_t NewSize) = 0;
|
||||
bool AllocateMoreSlots();
|
||||
};
|
||||
|
||||
} // namespace FEX::Profiler
|
||||
@@ -291,7 +291,7 @@ static bool TestInstructions(FEXCore::Context::Context* CTX, FEXCore::Core::Inte
|
||||
bool ShouldShowCode = INSTStats->first.HostCodeInstructions != CurrentTest->ExpectedInstructionCount;
|
||||
|
||||
if (ShouldShowCode) {
|
||||
for (auto Line : INSTStats->second) {
|
||||
for (const auto& Line : INSTStats->second) {
|
||||
LogMan::Msg::EFmt("\t{}", Line);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,150 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <imgui.h>
|
||||
#define YES_IMGUIFILESYSTEM
|
||||
#include <addons/imgui_user.h>
|
||||
|
||||
#include <SDL.h>
|
||||
#include <SDL_scancode.h>
|
||||
|
||||
#include <epoxy/gl.h>
|
||||
|
||||
#include "imgui_impl_sdl.h"
|
||||
#include "imgui_impl_opengl3.h"
|
||||
|
||||
#include <tuple>
|
||||
|
||||
namespace FEX::GUI {
|
||||
|
||||
using TupleReturn = std::tuple<SDL_Window*, SDL_GLContext>;
|
||||
static TupleReturn SetupIMGui(const char* Name, const fextl::string& Config) {
|
||||
// Setup SDL
|
||||
if (SDL_Init(SDL_INIT_VIDEO | SDL_INIT_TIMER | SDL_INIT_GAMECONTROLLER) != 0) {
|
||||
printf("Error: %s\n", SDL_GetError());
|
||||
return TupleReturn {};
|
||||
}
|
||||
|
||||
// Create window with graphics context
|
||||
SDL_GL_SetAttribute(SDL_GL_DOUBLEBUFFER, 1);
|
||||
SDL_GL_SetAttribute(SDL_GL_DEPTH_SIZE, 24);
|
||||
SDL_GL_SetAttribute(SDL_GL_STENCIL_SIZE, 8);
|
||||
SDL_WindowFlags window_flags = (SDL_WindowFlags)(SDL_WINDOW_OPENGL | SDL_WINDOW_RESIZABLE | SDL_WINDOW_ALLOW_HIGHDPI);
|
||||
SDL_Window* window = SDL_CreateWindow(Name, SDL_WINDOWPOS_CENTERED, SDL_WINDOWPOS_CENTERED, 640, 640, window_flags);
|
||||
SDL_GLContext gl_context {};
|
||||
const char* glsl_version {};
|
||||
|
||||
// Try a GL 3.0 context
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_FLAGS, 0);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_PROFILE_MASK, SDL_GL_CONTEXT_PROFILE_CORE);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MAJOR_VERSION, 3);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MINOR_VERSION, 0);
|
||||
gl_context = SDL_GL_CreateContext(window);
|
||||
glsl_version = "#version 130";
|
||||
|
||||
if (!gl_context) {
|
||||
// 3.0 failed, let's try 2.1
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_FLAGS, 0);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_PROFILE_MASK, SDL_GL_CONTEXT_PROFILE_COMPATIBILITY);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MAJOR_VERSION, 2);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MINOR_VERSION, 1);
|
||||
gl_context = SDL_GL_CreateContext(window);
|
||||
glsl_version = "#version 120";
|
||||
|
||||
if (!gl_context) {
|
||||
// 2.1 failed, let's try ES 2.0
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_FLAGS, 0);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_PROFILE_MASK, SDL_GL_CONTEXT_PROFILE_ES);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MAJOR_VERSION, 2);
|
||||
SDL_GL_SetAttribute(SDL_GL_CONTEXT_MINOR_VERSION, 0);
|
||||
gl_context = SDL_GL_CreateContext(window);
|
||||
glsl_version = "#version 100";
|
||||
|
||||
if (!gl_context) {
|
||||
printf("Couldn't create GL context: %s\n", SDL_GetError());
|
||||
return {};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SDL_GL_MakeCurrent(window, gl_context);
|
||||
SDL_GL_SetSwapInterval(1); // Enable vsync
|
||||
|
||||
// Setup Dear ImGui context
|
||||
IMGUI_CHECKVERSION();
|
||||
ImGui::CreateContext();
|
||||
ImGuiIO& io = ImGui::GetIO();
|
||||
io.ConfigFlags |= ImGuiConfigFlags_NavEnableKeyboard; // Enable Keyboard Controls
|
||||
io.ConfigFlags |= ImGuiConfigFlags_DockingEnable; // Enable Docking
|
||||
io.ConfigFlags |= ImGuiConfigFlags_ViewportsEnable; // Enable Multi-Viewport / Platform Windows
|
||||
io.IniFilename = &Config.at(0);
|
||||
|
||||
ImGui_ImplSDL2_InitForOpenGL(window, gl_context);
|
||||
ImGui_ImplOpenGL3_Init(glsl_version);
|
||||
return std::make_tuple(window, gl_context);
|
||||
}
|
||||
|
||||
static std::chrono::time_point<std::chrono::high_resolution_clock> LastUpdate {};
|
||||
constexpr auto UpdateTimeout = std::chrono::seconds(2);
|
||||
void DrawUI(SDL_Window* window, std::function<bool()> DrawFunction) {
|
||||
bool Running {true};
|
||||
ImGuiIO& io = ImGui::GetIO();
|
||||
while (Running) {
|
||||
SDL_Event event;
|
||||
auto Now = std::chrono::high_resolution_clock::now();
|
||||
auto Dur = Now - LastUpdate;
|
||||
|
||||
if (Dur < UpdateTimeout || SDL_WaitEvent(nullptr)) {
|
||||
while (SDL_PollEvent(&event)) {
|
||||
ImGui_ImplSDL2_ProcessEvent(&event);
|
||||
if (event.type == SDL_QUIT) {
|
||||
Running = false;
|
||||
}
|
||||
if (event.type == SDL_WINDOWEVENT && event.window.event == SDL_WINDOWEVENT_CLOSE && event.window.windowID == SDL_GetWindowID(window)) {
|
||||
Running = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Start the Dear ImGui frame
|
||||
ImGui_ImplOpenGL3_NewFrame();
|
||||
ImGui_ImplSDL2_NewFrame(window);
|
||||
Running &= DrawFunction();
|
||||
|
||||
glClear(GL_COLOR_BUFFER_BIT);
|
||||
ImGui_ImplOpenGL3_RenderDrawData(ImGui::GetDrawData());
|
||||
|
||||
if (io.ConfigFlags & ImGuiConfigFlags_ViewportsEnable) {
|
||||
SDL_Window* backup_current_window = SDL_GL_GetCurrentWindow();
|
||||
SDL_GLContext backup_current_context = SDL_GL_GetCurrentContext();
|
||||
ImGui::UpdatePlatformWindows();
|
||||
ImGui::RenderPlatformWindowsDefault();
|
||||
SDL_GL_MakeCurrent(backup_current_window, backup_current_context);
|
||||
}
|
||||
|
||||
SDL_GL_SwapWindow(window);
|
||||
}
|
||||
}
|
||||
|
||||
void Shutdown(SDL_Window* window, SDL_GLContext gl_context) {
|
||||
ImGui_ImplOpenGL3_Shutdown();
|
||||
ImGui_ImplSDL2_Shutdown();
|
||||
ImGui::DestroyContext();
|
||||
|
||||
SDL_GL_DeleteContext(gl_context);
|
||||
SDL_DestroyWindow(window);
|
||||
SDL_Quit();
|
||||
}
|
||||
|
||||
void HadUpdate() {
|
||||
LastUpdate = std::chrono::high_resolution_clock::now();
|
||||
|
||||
// Update the window
|
||||
SDL_Event Event {};
|
||||
Event.type = SDL_USEREVENT;
|
||||
SDL_PushEvent(&Event);
|
||||
}
|
||||
} // namespace FEX::GUI
|
||||
@@ -6,7 +6,6 @@ set(SRCS
|
||||
if (NOT MINGW_BUILD)
|
||||
list(APPEND SRCS
|
||||
Linux/Utils/ELFContainer.cpp
|
||||
Linux/Utils/ELFSymbolDatabase.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -401,7 +401,8 @@ public:
|
||||
}
|
||||
|
||||
uint64_t StackSize() const override {
|
||||
return sysconf(_SC_PAGESIZE);
|
||||
const auto Page = sysconf(_SC_PAGESIZE);
|
||||
return Page > 0 ? Page : FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
uint64_t GetStackPointer() override {
|
||||
@@ -426,7 +427,11 @@ public:
|
||||
return Result;
|
||||
};
|
||||
|
||||
const auto AllocPageSize = sysconf(_SC_PAGESIZE);
|
||||
auto AllocPageSize = sysconf(_SC_PAGESIZE);
|
||||
if (AllocPageSize <= 0) {
|
||||
AllocPageSize = FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
if (LimitedSize) {
|
||||
DoMMap(0xe000'0000, AllocPageSize * 10);
|
||||
|
||||
|
||||
@@ -1,361 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|elf-parsing
|
||||
desc: Part of our now defunct ld-linux replacement, keeps tracks of all symbols, loads elfs, handles relocations. Small parts of this are
|
||||
used. $end_info$
|
||||
*/
|
||||
|
||||
#include "Linux/Utils/ELFSymbolDatabase.h"
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <elf.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <tuple>
|
||||
|
||||
namespace ELFLoader {
|
||||
void ELFSymbolDatabase::FillLibrarySearchPaths() {
|
||||
// XXX: Open /etc/ld.so.conf and parse the paths in that file
|
||||
// For now I'm just filling with regular search paths
|
||||
if (File->GetMode() == ELFContainer::MODE_64BIT) {
|
||||
LibrarySearchPaths.emplace_back("/usr/local/lib/x86_64-linux-gnu");
|
||||
LibrarySearchPaths.emplace_back("/lib/x86_64-linux-gnu");
|
||||
LibrarySearchPaths.emplace_back("/usr/lib/x86_64-linux-gnu");
|
||||
} else {
|
||||
LibrarySearchPaths.emplace_back("/usr/local/lib/i386-linux-gnu");
|
||||
LibrarySearchPaths.emplace_back("/lib/i386-linux-gnu");
|
||||
LibrarySearchPaths.emplace_back("/usr/lib/i386-linux-gnu");
|
||||
}
|
||||
|
||||
// At least we can scan LD_LIBRARY_PATH
|
||||
auto EnvVar = getenv("LD_LIBRARY_PATH");
|
||||
if (EnvVar) {
|
||||
fextl::string Env = EnvVar;
|
||||
fextl::stringstream EnvStream(Env);
|
||||
fextl::string Token;
|
||||
while (std::getline(EnvStream, Token, ';')) {
|
||||
LibrarySearchPaths.emplace_back(Token);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool ELFSymbolDatabase::FindLibraryFile(fextl::string* Result, const char* Library) {
|
||||
for (auto& Path : LibrarySearchPaths) {
|
||||
const fextl::string TmpPath = fextl::fmt::format("{}/{}", Path, Library);
|
||||
if (FHU::Filesystem::Exists(TmpPath)) {
|
||||
*Result = TmpPath;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
ELFSymbolDatabase::ELFSymbolDatabase(::ELFLoader::ELFContainer* file)
|
||||
: File {file} {
|
||||
FillLibrarySearchPaths();
|
||||
|
||||
// Don't try to load the file if it was invalid
|
||||
if (!File->WasLoaded()) {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::vector<fextl::string> UnfilledDependencies;
|
||||
fextl::vector<ELFInfo*> NewLibraries;
|
||||
|
||||
auto FillDependencies = [&UnfilledDependencies, this](ELFInfo* ELF) {
|
||||
for (auto& Lib : *ELF->Container->GetNecessaryLibs()) {
|
||||
if (NameToELF.find(Lib) == NameToELF.end()) {
|
||||
UnfilledDependencies.emplace_back(Lib);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
auto LoadDependencies = [&UnfilledDependencies, &NewLibraries, this]() {
|
||||
for (auto& Lib : UnfilledDependencies) {
|
||||
if (NameToELF.find(Lib) == NameToELF.end()) {
|
||||
fextl::string LibraryPath;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
bool Found =
|
||||
#endif
|
||||
FindLibraryFile(&LibraryPath, Lib.c_str());
|
||||
LOGMAN_THROW_A_FMT(Found, "Couldn't find library '{}'", Lib);
|
||||
auto Info = DynamicELFInfo.emplace_back(new ELFInfo {});
|
||||
Info->Name = Lib;
|
||||
Info->Container = new ::ELFLoader::ELFContainer(LibraryPath, {}, true);
|
||||
NewLibraries.emplace_back(Info);
|
||||
NameToELF[Lib] = Info;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
LocalInfo.Container = File;
|
||||
LocalInfo.Name = "/proc/self/exe";
|
||||
NewLibraries.emplace_back(&LocalInfo);
|
||||
|
||||
do {
|
||||
fextl::vector<ELFInfo*> PreviousLibs;
|
||||
PreviousLibs.swap(NewLibraries);
|
||||
for (auto ELF : PreviousLibs) {
|
||||
FillDependencies(ELF);
|
||||
LoadDependencies();
|
||||
}
|
||||
} while (!UnfilledDependencies.empty() && !NewLibraries.empty());
|
||||
|
||||
FillMemoryLayouts(0);
|
||||
FillInitializationOrder();
|
||||
FillSymbols();
|
||||
|
||||
if (LocalInfo.Container->WasDynamic() && File->GetMode() == ELFContainer::MODE_64BIT) {
|
||||
ELFBase = FEXCore::Allocator::mmap(nullptr, ELFMemorySize, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
FillMemoryLayouts(reinterpret_cast<uintptr_t>(ELFBase));
|
||||
FillInitializationOrder();
|
||||
FillSymbols();
|
||||
|
||||
FixedNoReplace = false;
|
||||
}
|
||||
}
|
||||
|
||||
ELFSymbolDatabase::~ELFSymbolDatabase() {
|
||||
if (ELFBase) {
|
||||
FEXCore::Allocator::munmap(ELFBase, ELFMemorySize);
|
||||
ELFBase = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::FillMemoryLayouts(uint64_t DefinedBase) {
|
||||
uint64_t ELFBases = DefinedBase;
|
||||
ELFMemorySize = 0;
|
||||
if (!DefinedBase) {
|
||||
if (File->GetMode() == ELFContainer::MODE_64BIT) {
|
||||
ELFBases = 0x1'0000'0000;
|
||||
} else {
|
||||
// 32bit we will just load at the lowest memory address we can
|
||||
// Which on Linux is at 0x1'0000
|
||||
ELFBases = 0x1'0000;
|
||||
}
|
||||
}
|
||||
// We can only relocate the passed in ELF if it is dynamic
|
||||
// If it is EXEC then it HAS to end up in the base offset it chose
|
||||
if (LocalInfo.Container->WasDynamic()) {
|
||||
LocalInfo.CustomLayout = File->GetLayout();
|
||||
uint64_t CurrentELFBase = std::get<0>(LocalInfo.CustomLayout);
|
||||
uint64_t CurrentELFEnd = std::get<1>(LocalInfo.CustomLayout);
|
||||
uint64_t CurrentELFAlignedSize = FEXCore::AlignUp(std::get<2>(LocalInfo.CustomLayout), 4096);
|
||||
|
||||
CurrentELFBase += ELFBases;
|
||||
CurrentELFEnd += ELFBases;
|
||||
|
||||
std::get<0>(LocalInfo.CustomLayout) = CurrentELFBase;
|
||||
std::get<1>(LocalInfo.CustomLayout) = CurrentELFEnd;
|
||||
std::get<2>(LocalInfo.CustomLayout) = CurrentELFAlignedSize;
|
||||
LocalInfo.GuestBase = ELFBases;
|
||||
|
||||
ELFBases += CurrentELFAlignedSize;
|
||||
ELFMemorySize += CurrentELFAlignedSize;
|
||||
} else {
|
||||
LocalInfo.CustomLayout = File->GetLayout();
|
||||
uint64_t CurrentELFBase = std::get<0>(LocalInfo.CustomLayout);
|
||||
uint64_t CurrentELFAlignedSize = FEXCore::AlignUp(std::get<2>(LocalInfo.CustomLayout), 4096);
|
||||
if (CurrentELFBase < 0x10000) {
|
||||
// We can't allocate memory in the first 16KB, Hopefully no elfs require this.
|
||||
LOGMAN_MSG_A_FMT("Elf requires memory mapped in the first 16kb");
|
||||
}
|
||||
|
||||
std::get<2>(LocalInfo.CustomLayout) = CurrentELFAlignedSize;
|
||||
LocalInfo.GuestBase = 0;
|
||||
ELFMemorySize += CurrentELFAlignedSize;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < DynamicELFInfo.size(); ++i) {
|
||||
auto ELF = DynamicELFInfo[i]->Container;
|
||||
auto Layout = ELF->GetLayout();
|
||||
|
||||
uint64_t CurrentELFBase = std::get<0>(Layout);
|
||||
uint64_t CurrentELFEnd = std::get<1>(Layout);
|
||||
uint64_t CurrentELFAlignedSize = FEXCore::AlignUp(std::get<2>(Layout), 4096);
|
||||
|
||||
CurrentELFBase += ELFBases;
|
||||
CurrentELFEnd += ELFBases;
|
||||
|
||||
std::get<0>(Layout) = CurrentELFBase;
|
||||
std::get<1>(Layout) = CurrentELFEnd;
|
||||
std::get<2>(Layout) = CurrentELFAlignedSize;
|
||||
DynamicELFInfo[i]->CustomLayout = Layout;
|
||||
DynamicELFInfo[i]->GuestBase = ELFBases;
|
||||
|
||||
ELFBases += CurrentELFAlignedSize;
|
||||
ELFMemorySize += CurrentELFAlignedSize;
|
||||
}
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::FillInitializationOrder() {
|
||||
std::set<fextl::string> AlreadyInList;
|
||||
|
||||
while (true) {
|
||||
if (InitializationOrder.size() == DynamicELFInfo.size()) {
|
||||
break;
|
||||
}
|
||||
|
||||
for (auto& ELF : DynamicELFInfo) {
|
||||
// If this ELF is already in the list then skip it
|
||||
if (AlreadyInList.find(ELF->Name) != AlreadyInList.end()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
bool AllLibsLoaded = true;
|
||||
for (auto& Lib : *ELF->Container->GetNecessaryLibs()) {
|
||||
if (AlreadyInList.find(Lib) == AlreadyInList.end()) {
|
||||
AllLibsLoaded = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (AllLibsLoaded) {
|
||||
InitializationOrder.emplace_back(ELF);
|
||||
AlreadyInList.insert(ELF->Name);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::FillSymbols() {
|
||||
auto LocalSymbolFiller = [this](ELFLoader::ELFSymbol* Symbol) {
|
||||
Symbols.emplace_back(Symbol);
|
||||
Symbol->Address += LocalInfo.GuestBase;
|
||||
SymbolMap[Symbol->Name] = Symbol;
|
||||
SymbolMapByAddress[Symbol->Address] = Symbol;
|
||||
if (Symbol->Bind == STB_GLOBAL) {
|
||||
SymbolMapGlobalOnly[Symbol->Name] = Symbol;
|
||||
}
|
||||
if (Symbol->Bind != STB_WEAK) {
|
||||
SymbolMapNoWeak[Symbol->Name] = Symbol;
|
||||
}
|
||||
};
|
||||
|
||||
LocalInfo.Container->AddSymbols(LocalSymbolFiller);
|
||||
|
||||
// Let us fill symbols based on initialization order
|
||||
for (auto ELF : InitializationOrder) {
|
||||
auto SymbolFiller = [this, &ELF](ELFLoader::ELFSymbol* Symbol) {
|
||||
Symbols.emplace_back(Symbol);
|
||||
// Offset the address by the guest base
|
||||
Symbol->Address += ELF->GuestBase;
|
||||
SymbolMap[Symbol->Name] = Symbol;
|
||||
SymbolMapNoMain[Symbol->Name] = Symbol;
|
||||
SymbolMapByAddress[Symbol->Address] = Symbol;
|
||||
if (Symbol->Bind == STB_GLOBAL) {
|
||||
SymbolMapGlobalOnly[Symbol->Name] = Symbol;
|
||||
}
|
||||
if (Symbol->Bind != STB_WEAK) {
|
||||
SymbolMapNoWeak[Symbol->Name] = Symbol;
|
||||
SymbolMapNoMainNoWeak[Symbol->Name] = Symbol;
|
||||
}
|
||||
};
|
||||
|
||||
ELF->Container->AddSymbols(SymbolFiller);
|
||||
}
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::MapMemoryRegions(std::function<void*(uint64_t, uint64_t, bool)> Mapper) {
|
||||
auto Map = [&](ELFInfo& ELF) {
|
||||
uint64_t ELFBase = std::get<0>(ELF.CustomLayout);
|
||||
uint64_t ELFSize = std::get<2>(ELF.CustomLayout);
|
||||
uint64_t OffsetFromBase = ELFBase - ELF.GuestBase;
|
||||
ELF.ELFBase = static_cast<uint8_t*>(Mapper(ELFBase, ELFSize, FixedNoReplace)) - OffsetFromBase;
|
||||
};
|
||||
|
||||
Map(LocalInfo);
|
||||
for (auto* ELF : DynamicELFInfo) {
|
||||
Map(*ELF);
|
||||
}
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::WriteLoadableSections(::ELFLoader::ELFContainer::MemoryWriter Writer) {
|
||||
File->WriteLoadableSections(Writer, LocalInfo.GuestBase);
|
||||
|
||||
for (size_t i = 0; i < DynamicELFInfo.size(); ++i) {
|
||||
auto ELF = DynamicELFInfo[i]->Container;
|
||||
|
||||
ELF->WriteLoadableSections(Writer, DynamicELFInfo[i]->GuestBase);
|
||||
}
|
||||
|
||||
HandleRelocations();
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::HandleRelocations() {
|
||||
auto SymbolGetter = [this](const char* SymbolName, uint8_t Table) -> ELFLoader::ELFSymbol* {
|
||||
SymbolTableType& TablePtr = SymbolMap;
|
||||
if (Table == 0) {
|
||||
TablePtr = SymbolMap;
|
||||
} else if (Table == 1) { // Global
|
||||
TablePtr = SymbolMapGlobalOnly;
|
||||
} else if (Table == 2) { // NoWeak
|
||||
TablePtr = SymbolMapNoWeak;
|
||||
} else if (Table == 3) { // No Main
|
||||
TablePtr = SymbolMapNoMain;
|
||||
} else if (Table == 4) { // No Main No Weak
|
||||
TablePtr = SymbolMapNoMainNoWeak;
|
||||
}
|
||||
|
||||
auto Sym = TablePtr.find(SymbolName);
|
||||
if (Sym == TablePtr.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
return Sym->second;
|
||||
};
|
||||
|
||||
for (auto ELF : InitializationOrder) {
|
||||
ELF->Container->FixupRelocations(ELF->ELFBase, ELF->GuestBase, SymbolGetter);
|
||||
}
|
||||
|
||||
LocalInfo.Container->FixupRelocations(LocalInfo.ELFBase, LocalInfo.GuestBase, SymbolGetter);
|
||||
}
|
||||
|
||||
uint64_t ELFSymbolDatabase::GetElfBase() const {
|
||||
return LocalInfo.GuestBase;
|
||||
}
|
||||
|
||||
uint64_t ELFSymbolDatabase::DefaultRIP() const {
|
||||
return File->GetEntryPoint() + LocalInfo.GuestBase;
|
||||
}
|
||||
|
||||
const ELFSymbol* ELFSymbolDatabase::GetSymbolInRange(RangeType Address) {
|
||||
auto Sym = SymbolMapByAddress.upper_bound(Address.first);
|
||||
if (Sym != SymbolMapByAddress.begin()) {
|
||||
--Sym;
|
||||
}
|
||||
if (Sym == SymbolMapByAddress.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if ((Sym->second->Address + Sym->second->Size) < Address.first) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (Sym->second->Address > Address.first) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
return Sym->second;
|
||||
}
|
||||
|
||||
void ELFSymbolDatabase::GetInitLocations(fextl::vector<uint64_t>* Locations) {
|
||||
// Walk the initialization order and fill the locations for initializations
|
||||
for (auto ELF : InitializationOrder) {
|
||||
ELF->Container->GetInitLocations(ELF->GuestBase, Locations);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace ELFLoader
|
||||
@@ -1,76 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include "Linux/Utils/ELFContainer.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <utility>
|
||||
|
||||
namespace ELFLoader {
|
||||
class ELFSymbolDatabase final {
|
||||
public:
|
||||
ELFSymbolDatabase(::ELFLoader::ELFContainer* file);
|
||||
~ELFSymbolDatabase();
|
||||
|
||||
uint64_t GetElfBase() const;
|
||||
|
||||
void MapMemoryRegions(std::function<void*(uint64_t, uint64_t, bool)> Mapper);
|
||||
void WriteLoadableSections(::ELFLoader::ELFContainer::MemoryWriter Writer);
|
||||
|
||||
uint64_t DefaultRIP() const;
|
||||
|
||||
using RangeType = std::pair<uint64_t, uint64_t>;
|
||||
const ::ELFLoader::ELFSymbol* GetSymbolInRange(RangeType Address);
|
||||
const ::ELFLoader::ELFSymbol* GetGlobalSymbolInRange(RangeType Address);
|
||||
const ::ELFLoader::ELFSymbol* GetNoWeakSymbolInRange(RangeType Address);
|
||||
|
||||
void GetInitLocations(fextl::vector<uint64_t>* Locations);
|
||||
|
||||
private:
|
||||
::ELFLoader::ELFContainer* File;
|
||||
|
||||
struct ELFInfo {
|
||||
fextl::string Name;
|
||||
::ELFLoader::ELFContainer* Container;
|
||||
::ELFLoader::ELFContainer::MemoryLayout CustomLayout;
|
||||
void* ELFBase;
|
||||
uint64_t GuestBase;
|
||||
};
|
||||
|
||||
ELFInfo LocalInfo;
|
||||
fextl::vector<ELFInfo*> DynamicELFInfo;
|
||||
fextl::vector<ELFInfo*> InitializationOrder;
|
||||
uint64_t ELFMemorySize {};
|
||||
bool FixedNoReplace {true};
|
||||
void* ELFBase {};
|
||||
|
||||
fextl::unordered_map<fextl::string, ELFInfo*> NameToELF;
|
||||
fextl::vector<fextl::string> LibrarySearchPaths;
|
||||
|
||||
// Symbols
|
||||
fextl::vector<ELFLoader::ELFSymbol*> Symbols;
|
||||
using SymbolTableType = fextl::unordered_map<fextl::string, ELFLoader::ELFSymbol*>;
|
||||
SymbolTableType SymbolMap;
|
||||
SymbolTableType SymbolMapGlobalOnly;
|
||||
SymbolTableType SymbolMapNoWeak;
|
||||
SymbolTableType SymbolMapNoMain;
|
||||
SymbolTableType SymbolMapNoMainNoWeak;
|
||||
fextl::map<uint64_t, ELFLoader::ELFSymbol*> SymbolMapByAddress;
|
||||
|
||||
bool FindLibraryFile(fextl::string* Result, const char* Library);
|
||||
void FillLibrarySearchPaths();
|
||||
void FillMemoryLayouts(uint64_t DefinedBase);
|
||||
void FillInitializationOrder();
|
||||
void FillSymbols();
|
||||
|
||||
void HandleRelocations();
|
||||
|
||||
const ::ELFLoader::ELFSymbol* GetSymbolFromTable(RangeType Address, fextl::unordered_map<fextl::string, ELFLoader::ELFSymbol*>& Table);
|
||||
};
|
||||
} // namespace ELFLoader
|
||||
@@ -3,6 +3,19 @@
|
||||
#include "Common/Config.h"
|
||||
|
||||
namespace FEX {
|
||||
static inline std::optional<fextl::string> GetSelfPath() {
|
||||
// Read the FEXInterpreter path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
|
||||
// This way we can get the parent path that the application is executing from.
|
||||
char SelfPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
|
||||
if (Result == -1) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
std::string_view SelfPathView {SelfPath, std::min<size_t>(PATH_MAX, Result)};
|
||||
return fextl::string {SelfPathView.substr(0, SelfPathView.find_last_of('/') + 1)};
|
||||
}
|
||||
|
||||
static inline FEX::Config::PortableInformation ReadPortabilityInformation() {
|
||||
const FEX::Config::PortableInformation BadResult {false, {}};
|
||||
const char* PortableConfig = getenv("FEX_PORTABLE");
|
||||
@@ -17,17 +30,12 @@ static inline FEX::Config::PortableInformation ReadPortabilityInformation() {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
// Read the FEXInterpreter path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
|
||||
// This way we can get the parent path that the application is executing from.
|
||||
char SelfPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
|
||||
if (Result == -1) {
|
||||
auto SelfPath = GetSelfPath();
|
||||
if (!SelfPath) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
std::string_view SelfPathView {SelfPath, std::min<size_t>(PATH_MAX, Result)};
|
||||
|
||||
// Extract the absolute path from the FEXInterpreter path
|
||||
return {true, fextl::string {SelfPathView.substr(0, SelfPathView.find_last_of('/') + 1)}};
|
||||
return {true, *SelfPath};
|
||||
}
|
||||
} // namespace FEX
|
||||
+103
-33
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Main.h"
|
||||
|
||||
#include <Common/Async.h>
|
||||
#include <Common/Config.h>
|
||||
#include <Common/FileFormatCheck.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -27,6 +28,32 @@ static fextl::unique_ptr<FEXCore::Config::Layer> LoadedConfig {};
|
||||
static fextl::map<FEXCore::Config::ConfigOption, std::pair<std::string, std::string_view>> ConfigToNameLookup;
|
||||
static fextl::map<std::string, FEXCore::Config::ConfigOption> NameToConfigLookup;
|
||||
|
||||
#include "Common/JSONPool.h"
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
|
||||
static void LoadThunkDatabase(fextl::unordered_map<fextl::string, bool>& HostLibsDB, bool Global) {
|
||||
auto ThunkDBPath = FEXCore::Config::GetConfigDirectory(Global) + "ThunksDB.json";
|
||||
fextl::vector<char> FileData;
|
||||
if (!FEXCore::FileLoading::LoadFile(FileData, ThunkDBPath)) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(FileData, Pool);
|
||||
if (!json) {
|
||||
return;
|
||||
}
|
||||
|
||||
const json_t* DB = json_getProperty(json, "DB");
|
||||
if (!DB || JSON_OBJ != json_getType(DB)) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (const json_t* Library = json_getChild(DB); Library != nullptr; Library = json_getSibling(Library)) {
|
||||
HostLibsDB[json_getName(Library)] = false;
|
||||
}
|
||||
}
|
||||
|
||||
ConfigModel::ConfigModel() {
|
||||
setItemRoleNames(QHash<int, QByteArray> {{Qt::DisplayRole, "display"}, {Qt::UserRole + 1, "optionType"}, {Qt::UserRole + 2, "optionValue"}});
|
||||
Reload();
|
||||
@@ -187,8 +214,66 @@ static void ConfigInit(fextl::string ConfigFilename) {
|
||||
}
|
||||
}
|
||||
|
||||
HostLibsModel::HostLibsModel() {
|
||||
// Load list of available libraries
|
||||
LoadThunkDatabase(HostLibsDB, true);
|
||||
LoadThunkDatabase(HostLibsDB, false);
|
||||
}
|
||||
|
||||
void HostLibsModel::Reload(const fextl::string& Path) {
|
||||
for (auto& [_, Enabled] : HostLibsDB) {
|
||||
Enabled = false;
|
||||
}
|
||||
|
||||
{
|
||||
fextl::vector<char> FileData;
|
||||
if (!FEXCore::FileLoading::LoadFile(FileData, Path)) {
|
||||
goto RenderItems;
|
||||
}
|
||||
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(FileData, Pool);
|
||||
if (!json) {
|
||||
goto RenderItems;
|
||||
}
|
||||
|
||||
const json_t* ThunksDB = json_getProperty(json, "ThunksDB");
|
||||
if (!ThunksDB) {
|
||||
goto RenderItems;
|
||||
}
|
||||
|
||||
for (const json_t* Item = json_getChild(ThunksDB); Item != nullptr; Item = json_getSibling(Item)) {
|
||||
auto DBObject = HostLibsDB.find(json_getName(Item));
|
||||
if (DBObject != HostLibsDB.end()) {
|
||||
DBObject->second = (json_getInteger(Item) != 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RenderItems:
|
||||
beginResetModel();
|
||||
removeRows(0, rowCount());
|
||||
for (auto& [Name, Enabled] : HostLibsDB) {
|
||||
auto Item = new QStandardItem(QString::fromUtf8(Name.c_str()));
|
||||
Item->setData(Enabled, Qt::CheckStateRole);
|
||||
appendRow(Item);
|
||||
}
|
||||
endResetModel();
|
||||
}
|
||||
|
||||
QHash<int, QByteArray> HostLibsModel::roleNames() const {
|
||||
auto ret = QStandardItemModel::roleNames();
|
||||
ret[Qt::CheckStateRole] = "checked";
|
||||
return ret;
|
||||
}
|
||||
|
||||
bool HostLibsModel::setData(const QModelIndex& index, const QVariant& value, int role) {
|
||||
std::next(HostLibsDB.begin(), index.row())->second = value.toBool();
|
||||
return QStandardItemModel::setData(index, value, role);
|
||||
}
|
||||
|
||||
RootFSModel::RootFSModel() {
|
||||
INotifyFD = inotify_init1(IN_NONBLOCK | IN_CLOEXEC);
|
||||
auto INotifyFD = inotify_init1(IN_NONBLOCK | IN_CLOEXEC);
|
||||
|
||||
fextl::string RootFS = FEXCore::Config::GetDataDirectory(false) + "RootFS/";
|
||||
int LocalFolderWD = inotify_add_watch(INotifyFD, RootFS.c_str(), IN_CREATE | IN_DELETE);
|
||||
@@ -196,7 +281,8 @@ RootFSModel::RootFSModel() {
|
||||
RootFS = FEXCore::Config::GetDataDirectory(true) + "RootFS/";
|
||||
int GlobalFolderWD = inotify_add_watch(INotifyFD, RootFS.c_str(), IN_CREATE | IN_DELETE);
|
||||
if (INotifyFD != -1 && (LocalFolderWD != -1 || GlobalFolderWD != -1)) {
|
||||
Thread = std::thread {&RootFSModel::INotifyThreadFunc, this};
|
||||
INotifyReactor.enable_async_stop();
|
||||
Thread = std::thread {&RootFSModel::INotifyThreadFunc, this, INotifyFD};
|
||||
} else {
|
||||
qWarning() << "Could not set up inotify. RootFS folder won't be monitored for changes.";
|
||||
}
|
||||
@@ -206,10 +292,7 @@ RootFSModel::RootFSModel() {
|
||||
}
|
||||
|
||||
RootFSModel::~RootFSModel() {
|
||||
close(INotifyFD);
|
||||
INotifyFD = -1;
|
||||
|
||||
ExitRequest.count_down();
|
||||
INotifyReactor.stop_async();
|
||||
Thread.join();
|
||||
}
|
||||
|
||||
@@ -261,39 +344,22 @@ QUrl RootFSModel::getBaseUrl() const {
|
||||
return QUrl::fromLocalFile(QString::fromStdString(FEXCore::Config::GetDataDirectory().c_str()) + "RootFS/");
|
||||
}
|
||||
|
||||
void RootFSModel::INotifyThreadFunc() {
|
||||
while (!ExitRequest.try_wait()) {
|
||||
void RootFSModel::INotifyThreadFunc(int INotifyFD) {
|
||||
fasio::posix_descriptor INotify {INotifyReactor, INotifyFD};
|
||||
|
||||
INotify.async_wait([this, INotifyFD](fasio::error ec) {
|
||||
// Spin through the events, we don't actually care what they are
|
||||
constexpr size_t DATA_SIZE = (16 * (sizeof(struct inotify_event) + NAME_MAX + 1));
|
||||
char buf[DATA_SIZE];
|
||||
int Ret {};
|
||||
struct pollfd watch_fd = {
|
||||
.fd = INotifyFD,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
};
|
||||
do {
|
||||
// 50 ms
|
||||
struct timespec tv {
|
||||
.tv_sec = 0, .tv_nsec = 50'000'000,
|
||||
};
|
||||
|
||||
// Reset revents.
|
||||
watch_fd.revents = 0;
|
||||
Ret = ppoll(&watch_fd, 1, &tv, nullptr);
|
||||
} while (Ret == 0 && INotifyFD != -1);
|
||||
|
||||
if (Ret == -1 || INotifyFD == -1) {
|
||||
// Just return on error
|
||||
return;
|
||||
}
|
||||
|
||||
// Spin through the events, we don't actually care what they are
|
||||
while (read(INotifyFD, buf, DATA_SIZE) > 0)
|
||||
;
|
||||
|
||||
// Queue update to the data model
|
||||
QMetaObject::invokeMethod(this, "Reload");
|
||||
}
|
||||
return fasio::post_callback::repeat;
|
||||
});
|
||||
|
||||
INotifyReactor.run();
|
||||
}
|
||||
|
||||
// Returns true on success
|
||||
@@ -333,7 +399,10 @@ static bool OpenFile(fextl::string Filename) {
|
||||
}
|
||||
|
||||
ConfigRuntime::ConfigRuntime(const QString& ConfigFilename) {
|
||||
HostLibs.Reload(ConfigFilename.toStdString().c_str());
|
||||
|
||||
qmlRegisterSingletonInstance<ConfigModel>("FEX.ConfigModel", 1, 0, "ConfigModel", &ConfigModelInst);
|
||||
qmlRegisterSingletonInstance<HostLibsModel>("FEX.HostLibsModel", 1, 0, "HostLibsModel", &HostLibs);
|
||||
qmlRegisterSingletonInstance<RootFSModel>("FEX.RootFSModel", 1, 0, "RootFSModel", &RootFSList);
|
||||
Engine.load(QUrl("qrc:/main.qml"));
|
||||
|
||||
@@ -353,7 +422,7 @@ ConfigRuntime::ConfigRuntime(const QString& ConfigFilename) {
|
||||
|
||||
void ConfigRuntime::onSave(const QUrl& Filename) {
|
||||
qInfo() << "Saving to" << Filename.toLocalFile().toStdString().c_str();
|
||||
FEX::Config::SaveLayerToJSON(Filename.toLocalFile().toStdString().c_str(), LoadedConfig.get());
|
||||
FEX::Config::SaveLayerToJSON(Filename.toLocalFile().toStdString().c_str(), LoadedConfig.get(), HostLibs.HostLibsDB);
|
||||
}
|
||||
|
||||
void ConfigRuntime::onLoad(const QUrl& Filename) {
|
||||
@@ -372,6 +441,7 @@ void ConfigRuntime::onLoad(const QUrl& Filename) {
|
||||
|
||||
ConfigModelInst.Reload();
|
||||
RootFSList.Reload();
|
||||
HostLibs.Reload(Filename.toLocalFile().toStdString().c_str());
|
||||
|
||||
QMetaObject::invokeMethod(Window, "refreshUI");
|
||||
}
|
||||
|
||||
@@ -1,8 +1,11 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <Common/Async.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <QStandardItemModel>
|
||||
#include <QQmlApplicationEngine>
|
||||
|
||||
#include <latch>
|
||||
#include <thread>
|
||||
|
||||
class QQuickWindow;
|
||||
@@ -32,17 +35,32 @@ public slots:
|
||||
void setInt(const QString&, int value);
|
||||
};
|
||||
|
||||
class HostLibsModel : public QStandardItemModel {
|
||||
Q_OBJECT
|
||||
QML_ELEMENT
|
||||
QML_SINGLETON
|
||||
|
||||
public:
|
||||
fextl::unordered_map<fextl::string, bool> HostLibsDB;
|
||||
|
||||
HostLibsModel();
|
||||
|
||||
QHash<int, QByteArray> roleNames() const override;
|
||||
|
||||
bool setData(const QModelIndex&, const QVariant&, int role) override;
|
||||
|
||||
void Reload(const fextl::string& Filename);
|
||||
};
|
||||
|
||||
class RootFSModel : public QStandardItemModel {
|
||||
Q_OBJECT
|
||||
QML_ELEMENT
|
||||
QML_SINGLETON
|
||||
|
||||
std::thread Thread;
|
||||
std::latch ExitRequest {1};
|
||||
fasio::poll_reactor INotifyReactor;
|
||||
|
||||
int INotifyFD;
|
||||
|
||||
void INotifyThreadFunc();
|
||||
void INotifyThreadFunc(int INotifyFD);
|
||||
|
||||
public:
|
||||
RootFSModel();
|
||||
@@ -63,6 +81,7 @@ class ConfigRuntime : public QObject {
|
||||
QQuickWindow* Window = nullptr;
|
||||
RootFSModel RootFSList;
|
||||
ConfigModel ConfigModelInst;
|
||||
HostLibsModel HostLibs;
|
||||
|
||||
public:
|
||||
ConfigRuntime(const QString& ConfigFilename);
|
||||
|
||||
@@ -4,6 +4,7 @@ import QtQuick.Controls 2.15
|
||||
import QtQuick.Layouts 1.15
|
||||
|
||||
import FEX.ConfigModel 1.0
|
||||
import FEX.HostLibsModel 1.0
|
||||
import FEX.RootFSModel 1.0
|
||||
|
||||
// Qt 6 changed the API of the Dialogs module slightly.
|
||||
@@ -186,6 +187,8 @@ ApplicationWindow {
|
||||
id: tabBar
|
||||
currentIndex: 0
|
||||
|
||||
readonly property int advancedIndex: 4
|
||||
|
||||
TabButton {
|
||||
text: qsTr("General")
|
||||
}
|
||||
@@ -195,6 +198,9 @@ ApplicationWindow {
|
||||
TabButton {
|
||||
text: qsTr("CPU")
|
||||
}
|
||||
TabButton {
|
||||
text: qsTr("Libraries")
|
||||
}
|
||||
TabButton {
|
||||
text: qsTr("Advanced")
|
||||
}
|
||||
@@ -421,53 +427,6 @@ ApplicationWindow {
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
title: qsTr("Library forwarding:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
id: libfwdConfig
|
||||
|
||||
property url configDir: (() => {
|
||||
var configPath = urlToLocalFile(configFilename)
|
||||
var slashIndex = configPath.lastIndexOf('/')
|
||||
if (slashIndex === -1) {
|
||||
return ""
|
||||
}
|
||||
return "file://" + configPath.substr(0, slashIndex)
|
||||
})()
|
||||
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Configuration:")
|
||||
config: "ThunkConfig"
|
||||
dialog: FileDialog {
|
||||
title: qsTr("Select library forwarding configuration")
|
||||
nameFilters: [ qsTr("JSON files (*.json)"), qsTr("All files(*)") ]
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Host library folder:")
|
||||
config: "ThunkHostLibs"
|
||||
dialog: FolderDialog {
|
||||
title: qsTr("Select path for host libraries")
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Guest library folder:")
|
||||
config: "ThunkGuestLibs"
|
||||
dialog: FolderDialog {
|
||||
title: qsTr("Select path for guest libraries")
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
title: qsTr("Logging:")
|
||||
width: parent.width - parent.padding * 2
|
||||
@@ -833,11 +792,91 @@ ApplicationWindow {
|
||||
}
|
||||
}
|
||||
|
||||
// Libraries settings
|
||||
ScrollablePage {
|
||||
GroupBox {
|
||||
title: qsTr("Library forwarding:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
id: libfwdConfig
|
||||
|
||||
property url configDir: (() => {
|
||||
var configPath = urlToLocalFile(configFilename)
|
||||
var slashIndex = configPath.lastIndexOf('/')
|
||||
if (slashIndex === -1) {
|
||||
return ""
|
||||
}
|
||||
return "file://" + configPath.substr(0, slashIndex)
|
||||
})()
|
||||
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Host library folder:")
|
||||
config: "ThunkHostLibs"
|
||||
dialog: FolderDialog {
|
||||
title: qsTr("Select path for host libraries")
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Guest library folder:")
|
||||
config: "ThunkGuestLibs"
|
||||
dialog: FolderDialog {
|
||||
title: qsTr("Select path for guest libraries")
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
id: hostLibsGroupBox
|
||||
title: qsTr("Use host library for:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
width: hostLibsGroupBox.width - hostLibsGroupBox.padding * 2
|
||||
ScrollView {
|
||||
Layout.fillWidth: true
|
||||
Layout.maximumHeight: 200
|
||||
clip: true
|
||||
|
||||
Column {
|
||||
property string selectedItem
|
||||
property string explicitEntry
|
||||
|
||||
spacing: 4
|
||||
|
||||
Component.onCompleted: {
|
||||
// root.refreshCacheChanged.connect(initState)
|
||||
}
|
||||
|
||||
Repeater {
|
||||
model: HostLibsModel
|
||||
delegate: CheckBox {
|
||||
text: model.display
|
||||
visible: text !== "fex_thunk_test" // Hide test library
|
||||
checked: (root.refreshCache, model.checked)
|
||||
onToggled: {
|
||||
configDirty = true
|
||||
model.checked = checked
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Advanced settings
|
||||
// NOTE: This is wrapped in a Loader that dynamically instantiates/destroys the page contents whenever the tab is selected.
|
||||
// This avoids costly UI updates for its UI elements.
|
||||
// TODO: Options contained multiple times in JSON aren't listed (neither are they in old FEXConfig though)
|
||||
Loader { sourceComponent: tabBar.currentIndex === 3 ? advancedSettingsPage : null }
|
||||
Loader { sourceComponent: tabBar.currentIndex === tabBar.advancedIndex ? advancedSettingsPage : null }
|
||||
Component {
|
||||
id: advancedSettingsPage
|
||||
ScrollablePage {
|
||||
|
||||
@@ -4,19 +4,13 @@
|
||||
|
||||
#include "ArchHelpers/UContext.h"
|
||||
#include "CodeLoader.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FDUtils.h"
|
||||
#include "FEXCore/Utils/Allocator.h"
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
#include "VDSO_Emulation.h"
|
||||
#include "Linux/Utils/ELFParser.h"
|
||||
#include "Linux/Utils/ELFSymbolDatabase.h"
|
||||
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <cassert>
|
||||
#include <cstring>
|
||||
#include <random>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
@@ -564,21 +558,21 @@ public:
|
||||
// All done
|
||||
|
||||
// Setup AuxVars
|
||||
AuxVariables.emplace_back(auxv_t {11, getauxval(AT_UID)}); // AT_UID
|
||||
AuxVariables.emplace_back(auxv_t {12, getauxval(AT_EUID)}); // AT_EUID
|
||||
AuxVariables.emplace_back(auxv_t {13, getauxval(AT_GID)}); // AT_GID
|
||||
AuxVariables.emplace_back(auxv_t {14, getauxval(AT_EGID)}); // AT_EGID
|
||||
AuxVariables.emplace_back(auxv_t {17, getauxval(AT_CLKTCK)}); // AT_CLKTIK
|
||||
AuxVariables.emplace_back(auxv_t {6, 0x1000}); // AT_PAGESIZE
|
||||
AuxRandom = &AuxVariables.emplace_back(auxv_t {25, ~0ULL}); // AT_RANDOM
|
||||
AuxVariables.emplace_back(auxv_t {23, getauxval(AT_SECURE)}); // AT_SECURE
|
||||
AuxVariables.emplace_back(auxv_t {8, 0}); // AT_FLAGS
|
||||
AuxVariables.emplace_back(auxv_t {5, MainElf.phdrs.size()}); // AT_PHNUM
|
||||
AuxVariables.emplace_back(auxv_t {16, HWCap}); // AT_HWCAP
|
||||
AuxVariables.emplace_back(auxv_t {26, HWCap2}); // AT_HWCAP2
|
||||
AuxVariables.emplace_back(auxv_t {51, CalculateSignalStackSize()}); // AT_MINSIGSTKSZ
|
||||
AuxPlatform = &AuxVariables.emplace_back(auxv_t {24, ~0ULL}); // AT_PLATFORM
|
||||
AuxExecFN = &AuxVariables.emplace_back(auxv_t {AT_EXECFN, ~0ULL}); // AT_EXECFN
|
||||
AuxVariables.emplace_back(auxv_t {11, getauxval(AT_UID)}); // AT_UID
|
||||
AuxVariables.emplace_back(auxv_t {12, getauxval(AT_EUID)}); // AT_EUID
|
||||
AuxVariables.emplace_back(auxv_t {13, getauxval(AT_GID)}); // AT_GID
|
||||
AuxVariables.emplace_back(auxv_t {14, getauxval(AT_EGID)}); // AT_EGID
|
||||
AuxVariables.emplace_back(auxv_t {17, getauxval(AT_CLKTCK)}); // AT_CLKTIK
|
||||
AuxVariables.emplace_back(auxv_t {6, FEXCore::Utils::FEX_PAGE_SIZE}); // AT_PAGESIZE
|
||||
AuxRandom = &AuxVariables.emplace_back(auxv_t {25, ~0ULL}); // AT_RANDOM
|
||||
AuxVariables.emplace_back(auxv_t {23, getauxval(AT_SECURE)}); // AT_SECURE
|
||||
AuxVariables.emplace_back(auxv_t {8, 0}); // AT_FLAGS
|
||||
AuxVariables.emplace_back(auxv_t {5, MainElf.phdrs.size()}); // AT_PHNUM
|
||||
AuxVariables.emplace_back(auxv_t {16, HWCap}); // AT_HWCAP
|
||||
AuxVariables.emplace_back(auxv_t {26, HWCap2}); // AT_HWCAP2
|
||||
AuxVariables.emplace_back(auxv_t {51, CalculateSignalStackSize()}); // AT_MINSIGSTKSZ
|
||||
AuxPlatform = &AuxVariables.emplace_back(auxv_t {24, ~0ULL}); // AT_PLATFORM
|
||||
AuxExecFN = &AuxVariables.emplace_back(auxv_t {AT_EXECFN, ~0ULL}); // AT_EXECFN
|
||||
|
||||
if (Is64BitMode()) {
|
||||
AuxVariables.emplace_back(auxv_t {4, 0x38}); // AT_PHENT
|
||||
|
||||
@@ -340,7 +340,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
// Ensure FEXServer is setup before config options try to pull CONFIG_ROOTFS
|
||||
if (!FEXServerClient::SetupClient(argv[0])) {
|
||||
auto SelfPath = FEX::GetSelfPath();
|
||||
if (!FEXServerClient::SetupClient(SelfPath.value_or(argv[0]))) {
|
||||
LogMan::Msg::EFmt("FEXServerClient: Failure to setup client");
|
||||
return -1;
|
||||
}
|
||||
@@ -359,7 +360,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
LogMan::Throw::UnInstallHandler();
|
||||
LogMan::Msg::UnInstallHandler();
|
||||
} else {
|
||||
auto LogFile = OutputLog();
|
||||
const auto& LogFile = OutputLog();
|
||||
// If stderr or stdout then we need to dup the FD
|
||||
// In some cases some applications will close stderr and stdout
|
||||
// then redirect the FD to either a log OR some cases just not use
|
||||
@@ -388,7 +389,6 @@ int main(int argc, char** argv, char** const envp) {
|
||||
std::this_thread::sleep_for(std::chrono::seconds(StartupSleep()));
|
||||
}
|
||||
|
||||
FEXCore::Profiler::Init();
|
||||
FEXCore::Telemetry::Initialize();
|
||||
|
||||
if (!LDPath().empty() && Program.ProgramPath.starts_with(LDPath())) {
|
||||
@@ -492,6 +492,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
free(data);
|
||||
}
|
||||
|
||||
FEXCore::Profiler::Init(Program.ProgramName, Program.ProgramPath);
|
||||
|
||||
// System allocator is now system allocator or FEX
|
||||
FEXCore::Context::InitializeStaticTables(Loader.Is64BitMode() ? FEXCore::Context::MODE_64BIT : FEXCore::Context::MODE_32BIT);
|
||||
|
||||
|
||||
@@ -96,7 +96,7 @@ void ParseArguments(int argc, char** argv) {
|
||||
}
|
||||
|
||||
if (Options.is_set_by_user("force_ui")) {
|
||||
auto Option = Options["force_ui"];
|
||||
const auto& Option = Options["force_ui"];
|
||||
if (Option == "tty") {
|
||||
UIOption = UIOverrideOption::TTY;
|
||||
} else if (Option == "zenity") {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "ArgumentLoader.h"
|
||||
#include "Common/cpp-optparse/OptionParser.h"
|
||||
#include "PipeScanner.h"
|
||||
|
||||
#include "git_version.h"
|
||||
|
||||
@@ -20,6 +21,7 @@ FEXServerOptions Load(int argc, char** argv) {
|
||||
Parser.add_option("-p", "--persistent").action("store").type("int").set_default(0).set_optional_value(true).metavar("n").help("Make FEXServer persistent. Optional number of seconds");
|
||||
|
||||
Parser.add_option("-w", "--wait").action("store_true").set_default(false).help("Wait for the FEXServer to shutdown");
|
||||
Parser.add_option("--wait_pipe").action("store").type("int").set_default(-1).set_optional_value(true);
|
||||
|
||||
Parser.add_option("-v").action("version").help("Version string");
|
||||
|
||||
@@ -33,6 +35,10 @@ FEXServerOptions Load(int argc, char** argv) {
|
||||
}
|
||||
FEXOptions.PersistentTimeout = Options.get("persistent");
|
||||
|
||||
int WaitPipe = Options.get("wait_pipe");
|
||||
if (WaitPipe != -1) {
|
||||
PipeScanner::SetWaitPipe(WaitPipe);
|
||||
}
|
||||
return FEXOptions;
|
||||
}
|
||||
} // namespace FEXServer::Config
|
||||
@@ -1,9 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/FEXServerClient.h"
|
||||
#include <Common/Async.h>
|
||||
#include <Common/FEXServerClient.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
#include <poll.h>
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
@@ -12,12 +10,8 @@ void ClientMsgHandler(int FD, FEXServerClient::Logging::PacketMsg* const Msg, co
|
||||
}
|
||||
|
||||
namespace Logger {
|
||||
std::vector<struct pollfd> PollFDs {};
|
||||
std::mutex IncomingPollFDsLock {};
|
||||
std::vector<struct pollfd> IncomingPollFDs {};
|
||||
int LogClientQueuePipe[2];
|
||||
std::thread LogThread;
|
||||
std::atomic<bool> ShouldShutdown {false};
|
||||
std::atomic<int32_t> LoggerThreadTID {};
|
||||
|
||||
void HandleLogData(int Socket) {
|
||||
std::vector<uint8_t> Data(1500);
|
||||
@@ -32,6 +26,9 @@ void HandleLogData(int Socket) {
|
||||
// No more to read
|
||||
break;
|
||||
}
|
||||
} else if (Read == 0) {
|
||||
// Socket closed
|
||||
return;
|
||||
} else {
|
||||
if (errno == EWOULDBLOCK) {
|
||||
// no error
|
||||
@@ -58,71 +55,48 @@ void HandleLogData(int Socket) {
|
||||
}
|
||||
|
||||
void LogThreadFunc() {
|
||||
LoggerThreadTID = FHU::Syscalls::gettid();
|
||||
fasio::poll_reactor Reactor;
|
||||
|
||||
while (!ShouldShutdown) {
|
||||
struct timespec ts {};
|
||||
ts.tv_sec = 5;
|
||||
auto Pipe = fasio::posix_descriptor {Reactor, LogClientQueuePipe[0]};
|
||||
fextl::vector<fasio::posix_descriptor> Clients;
|
||||
|
||||
{
|
||||
std::unique_lock lk {IncomingPollFDsLock};
|
||||
PollFDs.insert(PollFDs.end(), std::make_move_iterator(IncomingPollFDs.begin()), std::make_move_iterator(IncomingPollFDs.end()));
|
||||
IncomingPollFDs.clear();
|
||||
// Wait for AppendLogFD to send file descriptors over LogClientQueuePipe.
|
||||
// When data becomes ready, we read the FD and register it to the reactor.
|
||||
Pipe.async_wait([&](fasio::error ec) {
|
||||
if (ec != fasio::error::success) {
|
||||
return fasio::post_callback::stop_reactor;
|
||||
}
|
||||
if (PollFDs.size() == 0) {
|
||||
pselect(0, nullptr, nullptr, nullptr, &ts, nullptr);
|
||||
} else {
|
||||
int Result = ppoll(&PollFDs.at(0), PollFDs.size(), &ts, nullptr);
|
||||
if (Result > 0) {
|
||||
// Walk the FDs and see if we got any results
|
||||
for (auto it = PollFDs.begin(); it != PollFDs.end();) {
|
||||
bool Erase {};
|
||||
if (it->revents != 0) {
|
||||
if (it->revents & POLLIN) {
|
||||
// Data from the socket
|
||||
HandleLogData(it->fd);
|
||||
} else if (it->revents & (POLLHUP | POLLERR | POLLNVAL | POLLRDHUP)) {
|
||||
// Error or hangup, close the socket and erase it from our list
|
||||
Erase = true;
|
||||
close(it->fd);
|
||||
}
|
||||
|
||||
it->revents = 0;
|
||||
--Result;
|
||||
}
|
||||
int ReceivedFD;
|
||||
read(Pipe.FD, &ReceivedFD, sizeof(ReceivedFD));
|
||||
|
||||
if (Erase) {
|
||||
it = PollFDs.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
|
||||
if (Result == 0) {
|
||||
// Early break if we've consumed all the results
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Register client and set up read callback
|
||||
Clients.emplace_back(Reactor, ReceivedFD);
|
||||
Clients.back().async_wait([&Clients, ReceivedFD](fasio::error ec) {
|
||||
if (ec != fasio::error::success) {
|
||||
std::iter_swap(std::find_if(Clients.begin(), Clients.end(), [=](auto& desc) { return desc.FD == ReceivedFD; }), std::prev(Clients.end()));
|
||||
Clients.pop_back();
|
||||
return fasio::post_callback::drop;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
HandleLogData(ReceivedFD);
|
||||
return fasio::post_callback::repeat;
|
||||
});
|
||||
|
||||
return fasio::post_callback::repeat;
|
||||
});
|
||||
|
||||
Reactor.run();
|
||||
}
|
||||
|
||||
void StartLogThread() {
|
||||
pipe2(LogClientQueuePipe, 0);
|
||||
|
||||
LogThread = std::thread(LogThreadFunc);
|
||||
}
|
||||
|
||||
void AppendLogFD(int FD) {
|
||||
{
|
||||
std::unique_lock lk {IncomingPollFDsLock};
|
||||
IncomingPollFDs.emplace_back(pollfd {
|
||||
.fd = FD,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
});
|
||||
}
|
||||
|
||||
// Wake up the thread immediately
|
||||
FHU::Syscalls::tgkill(::getpid(), LoggerThreadTID, SIGUSR1);
|
||||
write(LogClientQueuePipe[1], &FD, sizeof(FD));
|
||||
}
|
||||
|
||||
bool LogThreadRunning() {
|
||||
@@ -130,10 +104,7 @@ bool LogThreadRunning() {
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
ShouldShutdown = true;
|
||||
|
||||
// Wake up the thread immediately
|
||||
FHU::Syscalls::tgkill(::getpid(), LoggerThreadTID, SIGUSR1);
|
||||
close(LogClientQueuePipe[1]);
|
||||
|
||||
if (LogThread.joinable()) {
|
||||
LogThread.join();
|
||||
|
||||
@@ -109,10 +109,6 @@ int main(int argc, char** argv, char** const envp) {
|
||||
LogMan::Msg::InstallHandler(Logging::MsgHandler);
|
||||
}
|
||||
|
||||
// Scan for any incoming pipes
|
||||
// We will close these later
|
||||
PipeScanner::ScanForPipes();
|
||||
|
||||
if (!Options.Foreground) {
|
||||
DeparentSelf();
|
||||
}
|
||||
|
||||
@@ -1,42 +1,18 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <cstdlib>
|
||||
#include <dirent.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
#include <fcntl.h>
|
||||
|
||||
namespace PipeScanner {
|
||||
std::vector<int> IncomingPipes {};
|
||||
|
||||
// Scan and store any pipe files.
|
||||
// This will capture all pipe files so needs to be executed early.
|
||||
// This ensures we find any pipe files from execve for waiting FEXInterpreters.
|
||||
void ScanForPipes() {
|
||||
DIR* fd = opendir("/proc/self/fd");
|
||||
if (fd) {
|
||||
struct dirent* dir {};
|
||||
do {
|
||||
dir = readdir(fd);
|
||||
if (dir) {
|
||||
char* end {};
|
||||
int open_fd = std::strtol(dir->d_name, &end, 8);
|
||||
if (end != dir->d_name) {
|
||||
struct stat stat {};
|
||||
int result = fstat(open_fd, &stat);
|
||||
if (result == -1) {
|
||||
continue;
|
||||
}
|
||||
if (stat.st_mode & S_IFIFO) {
|
||||
// Close any incoming pipes
|
||||
IncomingPipes.emplace_back(open_fd);
|
||||
}
|
||||
}
|
||||
}
|
||||
} while (dir);
|
||||
|
||||
closedir(fd);
|
||||
}
|
||||
void SetWaitPipe(int FD) {
|
||||
int flags = fcntl(FD, F_GETFD);
|
||||
flags |= FD_CLOEXEC;
|
||||
fcntl(FD, F_SETFD, flags);
|
||||
IncomingPipes.emplace_back(FD);
|
||||
}
|
||||
|
||||
void ClosePipes() {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
namespace PipeScanner {
|
||||
void ScanForPipes();
|
||||
void SetWaitPipe(int FD);
|
||||
void ClosePipes();
|
||||
} // namespace PipeScanner
|
||||
@@ -3,9 +3,13 @@
|
||||
#include "Logger.h"
|
||||
#include "SquashFS.h"
|
||||
|
||||
#include "Common/FEXServerClient.h"
|
||||
#include <Common/AsyncNet.h>
|
||||
#include <Common/FEXServerClient.h>
|
||||
|
||||
#include <fmt/ranges.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cassert>
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <poll.h>
|
||||
@@ -18,9 +22,8 @@
|
||||
namespace ProcessPipe {
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
int ServerLockFD {-1};
|
||||
int ServerSocketFD {-1};
|
||||
int ServerFSSocketFD {-1};
|
||||
std::atomic<bool> ShouldShutdown {false};
|
||||
std::optional<fasio::tcp_acceptor> ServerAcceptor;
|
||||
std::optional<fasio::tcp_acceptor> ServerFSAcceptor;
|
||||
time_t RequestTimeout {10};
|
||||
bool Foreground {false};
|
||||
std::vector<struct pollfd> PollFDs {};
|
||||
@@ -176,161 +179,114 @@ bool InitializeServerPipe() {
|
||||
return true;
|
||||
}
|
||||
|
||||
static fasio::poll_reactor Reactor;
|
||||
|
||||
void HandleSocketData(fasio::tcp_socket&);
|
||||
|
||||
bool InitializeServerSocket(bool abstract) {
|
||||
|
||||
// Create the initial unix socket
|
||||
int fd = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (fd == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't create AF_UNIX socket: {} {}\n", errno, strerror(errno));
|
||||
return false;
|
||||
}
|
||||
|
||||
struct sockaddr_un addr {};
|
||||
addr.sun_family = AF_UNIX;
|
||||
|
||||
size_t SizeOfSocketString;
|
||||
fextl::string ServerSocketName;
|
||||
if (abstract) {
|
||||
auto ServerSocketName = FEXServerClient::GetServerSocketName();
|
||||
SizeOfSocketString = std::min(ServerSocketName.size() + 1, sizeof(addr.sun_path) - 1);
|
||||
addr.sun_path[0] = 0; // Abstract AF_UNIX sockets start with \0
|
||||
strncpy(addr.sun_path + 1, ServerSocketName.data(), SizeOfSocketString);
|
||||
ServerSocketName = FEXServerClient::GetServerSocketName();
|
||||
} else {
|
||||
auto ServerSocketPath = FEXServerClient::GetServerSocketPath();
|
||||
ServerSocketName = FEXServerClient::GetServerSocketPath();
|
||||
// Unlink the socket file if it exists
|
||||
// We are being asked to create a daemon, not error check
|
||||
// We don't care if this failed or not
|
||||
unlink(ServerSocketPath.c_str());
|
||||
|
||||
SizeOfSocketString = std::min(ServerSocketPath.size(), sizeof(addr.sun_path) - 1);
|
||||
strncpy(addr.sun_path, ServerSocketPath.data(), SizeOfSocketString);
|
||||
unlink(ServerSocketName.c_str());
|
||||
}
|
||||
// Include final null character.
|
||||
size_t SizeOfAddr = sizeof(addr.sun_family) + SizeOfSocketString;
|
||||
|
||||
// Bind the socket to the path
|
||||
int Result = bind(fd, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
|
||||
if (Result == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't bind AF_UNIX socket '{}': {} {}\n", addr.sun_path, errno, strerror(errno));
|
||||
close(fd);
|
||||
auto Acceptor = fasio::tcp_acceptor::create(Reactor, abstract, ServerSocketName);
|
||||
if (!Acceptor) {
|
||||
LogMan::Msg::EFmt("Failed to create FEXServer socket: error {} ({})", errno, strerror(errno));
|
||||
return false;
|
||||
}
|
||||
|
||||
listen(fd, 16);
|
||||
PollFDs.emplace_back(pollfd {
|
||||
.fd = fd,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
Acceptor->async_accept([](fasio::error ec, std::optional<fasio::tcp_socket> Socket) {
|
||||
if (ec != fasio::error::success) {
|
||||
if (ec == fasio::error::generic_errno) {
|
||||
LogMan::Msg::EFmt("FEXServer failed to establish client connection: error {} ({})", errno, strerror(errno));
|
||||
}
|
||||
// Ignore error and wait for next connection
|
||||
return fasio::post_callback::repeat;
|
||||
}
|
||||
|
||||
int FD = Socket->FD;
|
||||
Reactor.bind_handler(
|
||||
pollfd {
|
||||
.fd = FD,
|
||||
.events = POLLIN | POLLPRI | POLLRDHUP,
|
||||
.revents = 0,
|
||||
},
|
||||
[Socket = std::move(Socket).value()](fasio::error ec) mutable {
|
||||
if (ec != fasio::error::success) {
|
||||
close(Socket.FD);
|
||||
return fasio::post_callback::drop;
|
||||
}
|
||||
HandleSocketData(Socket);
|
||||
// Wait for next data
|
||||
return fasio::post_callback::repeat;
|
||||
});
|
||||
|
||||
// Wait for next connection
|
||||
return fasio::post_callback::repeat;
|
||||
});
|
||||
|
||||
if (abstract) {
|
||||
ServerSocketFD = fd;
|
||||
} else {
|
||||
ServerFSSocketFD = fd;
|
||||
}
|
||||
|
||||
(abstract ? ServerAcceptor : ServerFSAcceptor) = std::move(Acceptor).value();
|
||||
return true;
|
||||
}
|
||||
|
||||
void SendEmptyErrorPacket(int Socket) {
|
||||
void SendEmptyErrorPacket(fasio::tcp_socket& Socket) {
|
||||
FEXServerClient::FEXServerResultPacket Res {
|
||||
.Header {
|
||||
.Type = FEXServerClient::PacketType::TYPE_ERROR,
|
||||
},
|
||||
};
|
||||
|
||||
struct iovec iov {
|
||||
.iov_base = &Res, .iov_len = sizeof(Res),
|
||||
};
|
||||
|
||||
struct msghdr msg {
|
||||
.msg_name = nullptr, .msg_namelen = 0, .msg_iov = &iov, .msg_iovlen = 1,
|
||||
};
|
||||
|
||||
sendmsg(Socket, &msg, 0);
|
||||
fasio::mutable_buffer Data = {.Data = std::as_writable_bytes(std::span(&Res, 1))};
|
||||
fasio::error ec;
|
||||
write(Socket, Data, ec);
|
||||
}
|
||||
|
||||
void SendFDSuccessPacket(int Socket, int FD) {
|
||||
void SendFDSuccessPacket(fasio::tcp_socket& Socket, int FD) {
|
||||
FEXServerClient::FEXServerResultPacket Res {
|
||||
.Header {
|
||||
.Type = FEXServerClient::PacketType::TYPE_SUCCESS,
|
||||
},
|
||||
};
|
||||
|
||||
struct iovec iov {
|
||||
.iov_base = &Res, .iov_len = sizeof(Res),
|
||||
};
|
||||
|
||||
struct msghdr msg {
|
||||
.msg_name = nullptr, .msg_namelen = 0, .msg_iov = &iov, .msg_iovlen = 1,
|
||||
};
|
||||
|
||||
// Setup the ancillary buffer. This is where we will be getting pipe FDs
|
||||
// We only need 4 bytes for the FD
|
||||
constexpr size_t CMSG_SIZE = CMSG_SPACE(sizeof(int));
|
||||
union AncillaryBuffer {
|
||||
struct cmsghdr Header;
|
||||
uint8_t Buffer[CMSG_SIZE];
|
||||
};
|
||||
AncillaryBuffer AncBuf {};
|
||||
|
||||
// Now link to our ancilllary buffer
|
||||
msg.msg_control = AncBuf.Buffer;
|
||||
msg.msg_controllen = CMSG_SIZE;
|
||||
|
||||
// Now we need to setup the ancillary buffer data. We are only sending an FD
|
||||
struct cmsghdr* cmsg = CMSG_FIRSTHDR(&msg);
|
||||
cmsg->cmsg_len = CMSG_LEN(sizeof(int));
|
||||
cmsg->cmsg_level = SOL_SOCKET;
|
||||
cmsg->cmsg_type = SCM_RIGHTS;
|
||||
|
||||
// We are giving the daemon the write side of the pipe
|
||||
memcpy(CMSG_DATA(cmsg), &FD, sizeof(int));
|
||||
|
||||
sendmsg(Socket, &msg, 0);
|
||||
fasio::mutable_buffer Data = {.Data = std::as_writable_bytes(std::span(&Res, 1)), .FD = &FD};
|
||||
fasio::error ec;
|
||||
write(Socket, Data, ec);
|
||||
}
|
||||
|
||||
void HandleSocketData(int Socket) {
|
||||
void HandleSocketData(fasio::tcp_socket& Socket) {
|
||||
std::vector<uint8_t> Data(1500);
|
||||
size_t CurrentRead {};
|
||||
|
||||
// Get the current number of FDs of the process before we start handling sockets.
|
||||
GetMaxFDs();
|
||||
|
||||
while (true) {
|
||||
struct iovec iov {
|
||||
.iov_base = &Data.at(CurrentRead), .iov_len = Data.size() - CurrentRead,
|
||||
};
|
||||
fasio::mutable_buffer buffer = {std::as_writable_bytes(std::span(Data))};
|
||||
|
||||
struct msghdr msg {
|
||||
.msg_name = nullptr, .msg_namelen = 0, .msg_iov = &iov, .msg_iovlen = 1,
|
||||
};
|
||||
{
|
||||
fasio::error ec;
|
||||
|
||||
ssize_t Read = recvmsg(Socket, &msg, 0);
|
||||
if (Read <= msg.msg_iov->iov_len) {
|
||||
CurrentRead += Read;
|
||||
if (CurrentRead == Data.size()) {
|
||||
Data.resize(Data.size() << 1);
|
||||
} else {
|
||||
// No more to read
|
||||
break;
|
||||
}
|
||||
auto Read = Socket.read_some(buffer, ec);
|
||||
if (ec == fasio::error::success) {
|
||||
assert(Read >= sizeof(FEXServerClient::FEXServerRequestPacket));
|
||||
buffer = {buffer.Data.subspan(0, Read)};
|
||||
} else if (ec == fasio::error::eof) {
|
||||
return;
|
||||
} else {
|
||||
if (errno == EWOULDBLOCK) {
|
||||
// no error
|
||||
} else {
|
||||
perror("read");
|
||||
}
|
||||
break;
|
||||
perror("read");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
size_t CurrentOffset {};
|
||||
while (CurrentOffset < CurrentRead) {
|
||||
FEXServerClient::FEXServerRequestPacket* Req = reinterpret_cast<FEXServerClient::FEXServerRequestPacket*>(&Data[CurrentOffset]);
|
||||
while (buffer.size() > 0) {
|
||||
FEXServerClient::FEXServerRequestPacket* Req = reinterpret_cast<FEXServerClient::FEXServerRequestPacket*>(Data.data());
|
||||
switch (Req->Header.Type) {
|
||||
case FEXServerClient::PacketType::TYPE_KILL:
|
||||
ShouldShutdown = true;
|
||||
CurrentOffset += sizeof(FEXServerClient::FEXServerRequestPacket::BasicRequest);
|
||||
Reactor.stop_async();
|
||||
buffer += sizeof(FEXServerClient::FEXServerRequestPacket::BasicRequest);
|
||||
break;
|
||||
case FEXServerClient::PacketType::TYPE_GET_LOG_FD: {
|
||||
if (Logger::LogThreadRunning()) {
|
||||
@@ -353,7 +309,7 @@ void HandleSocketData(int Socket) {
|
||||
SendEmptyErrorPacket(Socket);
|
||||
}
|
||||
|
||||
CurrentOffset += sizeof(FEXServerClient::FEXServerRequestPacket::Header);
|
||||
buffer += sizeof(FEXServerClient::FEXServerRequestPacket::Header);
|
||||
break;
|
||||
}
|
||||
case FEXServerClient::PacketType::TYPE_GET_ROOTFS_PATH: {
|
||||
@@ -370,28 +326,15 @@ void HandleSocketData(int Socket) {
|
||||
|
||||
char Null {};
|
||||
|
||||
iovec iov[3] {
|
||||
{
|
||||
.iov_base = &Res,
|
||||
.iov_len = sizeof(Res),
|
||||
},
|
||||
{
|
||||
.iov_base = const_cast<char*>(MountFolder.data()),
|
||||
.iov_len = MountFolder.size(),
|
||||
},
|
||||
{
|
||||
.iov_base = &Null,
|
||||
.iov_len = 1,
|
||||
},
|
||||
fasio::mutable_buffer Data[] = {
|
||||
{.Data = std::as_writable_bytes(std::span(&Res, 1))},
|
||||
{.Data = std::as_writable_bytes(std::span(const_cast<fextl::string&>(MountFolder)))},
|
||||
{.Data = std::as_writable_bytes(std::span(&Null, 1))},
|
||||
};
|
||||
fasio::error ec;
|
||||
write(Socket, Chained(Data), ec);
|
||||
|
||||
struct msghdr msg {
|
||||
.msg_name = nullptr, .msg_namelen = 0, .msg_iov = iov, .msg_iovlen = 3,
|
||||
};
|
||||
|
||||
sendmsg(Socket, &msg, 0);
|
||||
|
||||
CurrentOffset += sizeof(FEXServerClient::FEXServerRequestPacket::BasicRequest);
|
||||
buffer += sizeof(FEXServerClient::FEXServerRequestPacket::BasicRequest);
|
||||
break;
|
||||
}
|
||||
case FEXServerClient::PacketType::TYPE_GET_PID_FD: {
|
||||
@@ -420,16 +363,16 @@ void HandleSocketData(int Socket) {
|
||||
close(FD);
|
||||
}
|
||||
|
||||
CurrentOffset += sizeof(FEXServerClient::FEXServerRequestPacket::Header);
|
||||
buffer += sizeof(FEXServerClient::FEXServerRequestPacket::Header);
|
||||
break;
|
||||
}
|
||||
// Invalid
|
||||
// Invalid
|
||||
case FEXServerClient::PacketType::TYPE_ERROR:
|
||||
default:
|
||||
// Something sent us an invalid packet. To ensure we don't spin infinitely, consume all the data.
|
||||
LogMan::Msg::EFmt("[FEXServer] InvalidPacket size received 0x{:x} bytes", CurrentRead - CurrentOffset);
|
||||
CurrentOffset = CurrentRead;
|
||||
break;
|
||||
// Something sent us an invalid packet. Drop this client and continue
|
||||
LogMan::Msg::EFmt("Invalid FEXServer packet received: {:02x}", fmt::join(buffer.Data, ""));
|
||||
close(Socket.FD);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -440,88 +383,15 @@ void CloseConnections() {
|
||||
close(ServerLockFD);
|
||||
|
||||
// Close the server socket so no more connections can be started
|
||||
close(ServerSocketFD);
|
||||
close(ServerFSSocketFD);
|
||||
ServerAcceptor.reset();
|
||||
ServerFSAcceptor.reset();
|
||||
}
|
||||
|
||||
void WaitForRequests() {
|
||||
auto LastDataTime = std::chrono::system_clock::now();
|
||||
Reactor.enable_async_stop();
|
||||
Reactor.run(Foreground ? std::nullopt : std::optional {std::chrono::seconds {RequestTimeout}});
|
||||
|
||||
while (!ShouldShutdown) {
|
||||
struct timespec ts {};
|
||||
ts.tv_sec = RequestTimeout;
|
||||
|
||||
int Result = ppoll(&PollFDs.at(0), PollFDs.size(), &ts, nullptr);
|
||||
std::vector<struct pollfd> NewPollFDs {};
|
||||
|
||||
if (Result > 0) {
|
||||
// Walk the FDs and see if we got any results
|
||||
for (auto it = PollFDs.begin(); it != PollFDs.end();) {
|
||||
auto& Event = *it;
|
||||
bool Erase {};
|
||||
|
||||
if (Event.revents != 0) {
|
||||
if (Event.fd == ServerSocketFD || Event.fd == ServerFSSocketFD) {
|
||||
if (Event.revents & POLLIN) {
|
||||
// If it is the listen socket then we have a new connection
|
||||
struct sockaddr_storage Addr {};
|
||||
socklen_t AddrSize {};
|
||||
int NewFD = accept(Event.fd, reinterpret_cast<struct sockaddr*>(&Addr), &AddrSize);
|
||||
|
||||
// Add the new client to the temporary array
|
||||
NewPollFDs.emplace_back(pollfd {
|
||||
.fd = NewFD,
|
||||
.events = POLLIN | POLLPRI | POLLRDHUP,
|
||||
.revents = 0,
|
||||
});
|
||||
} else if (Event.revents & (POLLHUP | POLLERR | POLLNVAL)) {
|
||||
// Listen socket error or shutting down
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
if (Event.revents & POLLIN) {
|
||||
// Data from the socket
|
||||
HandleSocketData(Event.fd);
|
||||
}
|
||||
|
||||
if (Event.revents & (POLLHUP | POLLERR | POLLNVAL | POLLRDHUP)) {
|
||||
// Error or hangup, close the socket and erase it from our list
|
||||
Erase = true;
|
||||
close(Event.fd);
|
||||
}
|
||||
}
|
||||
|
||||
Event.revents = 0;
|
||||
--Result;
|
||||
}
|
||||
|
||||
if (Erase) {
|
||||
it = PollFDs.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
|
||||
if (Result == 0) {
|
||||
// Early break if we've consumed all the results
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Insert the new FDs to poll
|
||||
PollFDs.insert(PollFDs.begin(), NewPollFDs.begin(), NewPollFDs.end());
|
||||
|
||||
LastDataTime = std::chrono::system_clock::now();
|
||||
} else {
|
||||
auto Now = std::chrono::system_clock::now();
|
||||
auto Diff = Now - LastDataTime;
|
||||
if (Diff >= std::chrono::seconds(RequestTimeout) && !Foreground && PollFDs.size() == 2) {
|
||||
// If we aren't running in the foreground and we have no connections after a timeout
|
||||
// Then we can just go ahead and leave
|
||||
ShouldShutdown = true;
|
||||
LogMan::Msg::DFmt("[FEXServer] Shutting Down");
|
||||
}
|
||||
}
|
||||
}
|
||||
LogMan::Msg::DFmt("[FEXServer] Shutting Down");
|
||||
|
||||
CloseConnections();
|
||||
}
|
||||
@@ -532,6 +402,6 @@ void SetConfiguration(bool Foreground, uint32_t PersistentTimeout) {
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
ShouldShutdown = true;
|
||||
Reactor.stop_async();
|
||||
}
|
||||
} // namespace ProcessPipe
|
||||
@@ -9,7 +9,6 @@ set (SRCS
|
||||
LinuxSyscalls/FaultSafeUserMemAccess.cpp
|
||||
LinuxSyscalls/FileManagement.cpp
|
||||
LinuxSyscalls/LinuxAllocator.cpp
|
||||
LinuxSyscalls/NetStream.cpp
|
||||
LinuxSyscalls/Seccomp/SeccompEmulator.cpp
|
||||
LinuxSyscalls/Seccomp/BPFEmitter.cpp
|
||||
LinuxSyscalls/Seccomp/Dumper.cpp
|
||||
|
||||
@@ -263,7 +263,7 @@ FileManager::FileManager(FEXCore::Context::Context* ctx)
|
||||
}
|
||||
|
||||
// Now that we loaded the thunks object, walk through and ensure dependencies are enabled as well
|
||||
auto ThunkGuestPath = Is64BitMode() ? ThunkGuestLibs() : ThunkGuestLibs32();
|
||||
const auto& ThunkGuestPath = Is64BitMode() ? ThunkGuestLibs() : ThunkGuestLibs32();
|
||||
for (const auto& DBObject : ThunkDB) {
|
||||
if (!DBObject.second.Enabled) {
|
||||
continue;
|
||||
@@ -325,12 +325,15 @@ FileManager::FileManager(FEXCore::Context::Context* ctx)
|
||||
|
||||
// Keep an fd open for /proc, to bypass chroot-style sandboxes
|
||||
ProcFD = open("/proc", O_RDONLY | O_CLOEXEC);
|
||||
|
||||
// Track the st_dev of /proc, to check for inode equality
|
||||
struct stat Buffer;
|
||||
auto Result = fstat(ProcFD, &Buffer);
|
||||
if (Result >= 0) {
|
||||
ProcFSDev = Buffer.st_dev;
|
||||
if (ProcFD != -1) {
|
||||
// Track the st_dev of /proc, to check for inode equality
|
||||
struct stat Buffer;
|
||||
auto Result = fstat(ProcFD, &Buffer);
|
||||
if (Result >= 0) {
|
||||
ProcFSDev = Buffer.st_dev;
|
||||
}
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Couldn't open `/proc`. Is ProcFS mounted? FEX won't be able to track FD conflicts");
|
||||
}
|
||||
|
||||
UpdatePID(::getpid());
|
||||
|
||||
@@ -9,8 +9,6 @@ $end_info$
|
||||
#include "CodeLoader.h"
|
||||
#include "GdbServer/Info.h"
|
||||
|
||||
#include "LinuxSyscalls/NetStream.h"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <iomanip>
|
||||
@@ -47,7 +45,6 @@ $end_info$
|
||||
#endif
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <poll.h>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <string_view>
|
||||
@@ -64,7 +61,7 @@ namespace FEX {
|
||||
#ifndef _WIN32
|
||||
void GdbServer::Break(FEXCore::Core::InternalThreadState* Thread, int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (!CommsStream.HasSocket()) {
|
||||
if (!CommsSocket) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -73,7 +70,7 @@ void GdbServer::Break(FEXCore::Core::InternalThreadState* Thread, int signal) {
|
||||
CurrentDebuggingThread = ThreadObject->ThreadInfo.TID.load();
|
||||
|
||||
const auto str = fextl::fmt::format("T{:02x}thread:{:x};", signal, CurrentDebuggingThread);
|
||||
SendPacket(str);
|
||||
SendPacket(*CommsSocket, str);
|
||||
}
|
||||
|
||||
void GdbServer::WaitForThreadWakeup() {
|
||||
@@ -83,7 +80,6 @@ void GdbServer::WaitForThreadWakeup() {
|
||||
|
||||
GdbServer::~GdbServer() {
|
||||
CloseListenSocket();
|
||||
CoreShuttingDown = true;
|
||||
|
||||
if (gdbServerThread->joinable()) {
|
||||
gdbServerThread->join(nullptr);
|
||||
@@ -178,7 +174,7 @@ static fextl::string encodeHex(std::string_view str) {
|
||||
// Takes a serial stream and reads a single packet
|
||||
// Un-escapes chars, checks the checksum and request a retransmit if it fails.
|
||||
// Once the checksum is validated, it acknowledges and returns the packet in a string
|
||||
fextl::string GdbServer::ReadPacket() {
|
||||
fextl::string GdbServer::ReadPacket(const std::span<std::byte>& RawMessage) {
|
||||
fextl::string packet {};
|
||||
|
||||
// The GDB "Remote Serial Protocal" was originally 7bit clean for use on serial ports.
|
||||
@@ -190,36 +186,36 @@ fextl::string GdbServer::ReadPacket() {
|
||||
// where any $ or # in the packet body are escaped ('}' followed by the char XORed with 0x20)
|
||||
// The checksum is a single unsigned byte sum of the data, hex encoded.
|
||||
|
||||
Utils::NetStream::ReturnGet c;
|
||||
while ((c = CommsStream.get()).HasData()) {
|
||||
switch (c.GetData()) {
|
||||
case '$': // start of packet
|
||||
if (packet.size() != 0) {
|
||||
LogMan::Msg::EFmt("Dropping unexpected data: \"{}\"", packet);
|
||||
}
|
||||
if (RawMessage.empty() || (char)RawMessage[0] != '$') {
|
||||
ERROR_AND_DIE_FMT("Expected GDB protocol messages to start with '$'");
|
||||
}
|
||||
|
||||
// clear any existing data, must have been a mistake.
|
||||
packet = fextl::string();
|
||||
for (auto It = std::next(RawMessage.begin()); It != RawMessage.end(); ++It) {
|
||||
char c = (char)*It;
|
||||
switch (c) {
|
||||
case '$': // start of packet
|
||||
ERROR_AND_DIE_FMT("Unescaped control character");
|
||||
break;
|
||||
|
||||
case '}': // escape char
|
||||
{
|
||||
Utils::NetStream::ReturnGet escaped;
|
||||
|
||||
do {
|
||||
escaped = CommsStream.get();
|
||||
} while (!escaped.HasData() && !escaped.HasHangup());
|
||||
|
||||
if (escaped.HasData()) {
|
||||
packet.push_back(escaped.GetData() ^ 0x20);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Received Invalid escape char: ${}", packet);
|
||||
if (std::next(It) == RawMessage.end()) {
|
||||
ERROR_AND_DIE_FMT("Missing character after escape indicator");
|
||||
}
|
||||
char escaped = (char)*++It;
|
||||
packet.push_back(escaped ^ 0x20);
|
||||
break;
|
||||
}
|
||||
|
||||
case '#': // end of packet
|
||||
{
|
||||
if (RawMessage.end() - It <= 2) {
|
||||
ERROR_AND_DIE_FMT("Missing checksum at end of packet");
|
||||
}
|
||||
|
||||
char hexString[3] = {0, 0, 0};
|
||||
CommsStream.read(hexString, 2, true);
|
||||
hexString[0] = (char)*++It;
|
||||
hexString[1] = (char)*++It;
|
||||
int expected_checksum = std::strtoul(hexString, nullptr, 16);
|
||||
|
||||
if (calculateChecksum(packet) == expected_checksum) {
|
||||
@@ -229,7 +225,8 @@ fextl::string GdbServer::ReadPacket() {
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: packet.push_back(c.GetData()); break;
|
||||
|
||||
default: packet.push_back(c); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -256,22 +253,25 @@ static fextl::string escapePacket(const fextl::string& packet) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
void GdbServer::SendPacket(const fextl::string& packet) {
|
||||
void GdbServer::SendPacket(fasio::tcp_socket& Socket, const fextl::string& packet) {
|
||||
const auto escaped = escapePacket(packet);
|
||||
const auto str = fextl::fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
auto str = fextl::fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
|
||||
CommsStream.SendPacket(str);
|
||||
fasio::error ec;
|
||||
write(Socket, fasio::mutable_buffer {std::as_writable_bytes(std::span {str})}, ec);
|
||||
}
|
||||
|
||||
void GdbServer::SendACK(bool NACK) {
|
||||
void GdbServer::SendACK(fasio::tcp_socket& Socket, bool NACK) {
|
||||
if (NoAckMode) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (NACK) {
|
||||
CommsStream.SendPacket("-");
|
||||
std::string_view message = "-";
|
||||
send(Socket.FD, message.data(), message.size(), 0);
|
||||
} else {
|
||||
CommsStream.SendPacket("+");
|
||||
std::string_view message = "+";
|
||||
send(Socket.FD, message.data(), message.size(), 0);
|
||||
}
|
||||
|
||||
if (SettingNoAckMode) {
|
||||
@@ -1349,104 +1349,133 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const fextl::string& packe
|
||||
void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_ACK || response.TypeResponse == HandledPacketType::TYPE_ONLYACK) {
|
||||
SendACK(false);
|
||||
SendACK(*CommsSocket, false);
|
||||
} else if (response.TypeResponse == HandledPacketType::TYPE_NACK || response.TypeResponse == HandledPacketType::TYPE_ONLYNACK) {
|
||||
SendACK(true);
|
||||
SendACK(*CommsSocket, true);
|
||||
}
|
||||
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
SendPacket("");
|
||||
SendPacket(*CommsSocket, "");
|
||||
} else if (response.TypeResponse != HandledPacketType::TYPE_ONLYNACK && response.TypeResponse != HandledPacketType::TYPE_ONLYACK &&
|
||||
response.TypeResponse != HandledPacketType::TYPE_NONE) {
|
||||
SendPacket(response.Response);
|
||||
SendPacket(*CommsSocket, response.Response);
|
||||
}
|
||||
}
|
||||
|
||||
GdbServer::WaitForConnectionResult GdbServer::WaitForConnection() {
|
||||
while (!CoreShuttingDown.load()) {
|
||||
struct pollfd PollFD {
|
||||
.fd = ListenSocket, .events = POLLIN | POLLPRI | POLLRDHUP, .revents = 0,
|
||||
};
|
||||
int Result = ppoll(&PollFD, 1, nullptr, nullptr);
|
||||
if (Result > 0) {
|
||||
if (PollFD.revents & POLLIN) {
|
||||
OpenSocket();
|
||||
return WaitForConnectionResult::CONNECTION;
|
||||
} else if (PollFD.revents & (POLLHUP | POLLERR | POLLNVAL)) {
|
||||
// Listen socket error or shutting down
|
||||
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}");
|
||||
return WaitForConnectionResult::ERROR;
|
||||
}
|
||||
} else if (Result == -1) {
|
||||
LogMan::Msg::EFmt("[GdbServer] poll failure: {}", errno);
|
||||
std::pair<fextl::vector<std::byte>::iterator, bool>
|
||||
GdbServer::MatchPacket(fextl::vector<std::byte>::iterator begin, fextl::vector<std::byte>::iterator end) {
|
||||
if (CommsBuffer.empty()) {
|
||||
return std::make_pair(begin, false);
|
||||
}
|
||||
switch ((char)CommsBuffer[0]) {
|
||||
case '+':
|
||||
case '-':
|
||||
case '\x03':
|
||||
// No further data
|
||||
return std::make_pair(std::next(begin), true);
|
||||
|
||||
case '$': {
|
||||
// Message format: $packet-data#checksum, where checksum is a single byte.
|
||||
auto match = std::find(begin, end, (std::byte)'#');
|
||||
if (match == end) {
|
||||
// No match; fetch more data
|
||||
return std::make_pair(end, false);
|
||||
} else if (end - match <= 2) {
|
||||
// Found '#' but missing the checksum bytes
|
||||
return std::make_pair(match, false);
|
||||
} else {
|
||||
return std::make_pair(std::next(match, 3), true);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
LogMan::Msg::EFmt("[GdbServer] Shutting Down");
|
||||
return WaitForConnectionResult::ERROR;
|
||||
default: ERROR_AND_DIE_FMT("Unexpected character at beginning of GDB packet: {}", CommsBuffer[0]);
|
||||
}
|
||||
}
|
||||
|
||||
void GdbServer::HandlePacket(fasio::error ec, size_t BytesInMessage) {
|
||||
if (ec != fasio::error::success || BytesInMessage == 0) {
|
||||
ERROR_AND_DIE_FMT("Failed");
|
||||
}
|
||||
|
||||
char c = (char)CommsBuffer[0];
|
||||
switch (c) {
|
||||
case '$': {
|
||||
auto packet = ReadPacket(std::span {CommsBuffer}.subspan(0, BytesInMessage));
|
||||
auto response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown packet {}", packet);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case '+':
|
||||
// ACK, do nothing.
|
||||
break;
|
||||
case '-':
|
||||
// NAK, Resend requested
|
||||
{
|
||||
std::lock_guard lk(sendMutex);
|
||||
SendPacket(*CommsSocket, {});
|
||||
}
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
SyscallHandler->TM.Pause();
|
||||
fextl::string str = fextl::fmt::format("T02thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
}
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("GdbServer: Unexpected byte {} ({:02x})", c, c);
|
||||
}
|
||||
|
||||
CommsBuffer.erase(CommsBuffer.begin(), CommsBuffer.begin() + BytesInMessage);
|
||||
|
||||
async_read_until(*CommsSocket, fasio::dynamic_vector_buffer {CommsBuffer}, std::bind_front(&GdbServer::MatchPacket, this),
|
||||
std::bind_front(&GdbServer::HandlePacket, this));
|
||||
}
|
||||
|
||||
|
||||
void GdbServer::GdbServerLoop() {
|
||||
OpenListenSocket();
|
||||
if (ListenSocket == -1) {
|
||||
if (!Acceptor) {
|
||||
// Couldn't open socket, just exit.
|
||||
return;
|
||||
}
|
||||
|
||||
while (!CoreShuttingDown.load()) {
|
||||
if (WaitForConnection() == WaitForConnectionResult::ERROR) {
|
||||
break;
|
||||
Acceptor->async_accept([this](fasio::error ec, std::optional<fasio::tcp_socket> Socket) {
|
||||
if (ec != fasio::error::success) {
|
||||
// Listen socket error or shutting down
|
||||
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}");
|
||||
close(CommsSocket->FD);
|
||||
CommsSocket.reset();
|
||||
// Repeat to wait for another connection
|
||||
return fasio::post_callback::repeat;
|
||||
}
|
||||
|
||||
HandledPacketType response {};
|
||||
CommsSocket.emplace(*std::move(Socket));
|
||||
|
||||
while (!CoreShuttingDown.load()) {
|
||||
// Outer server loop. Handles packet start, ACK/NAK and break
|
||||
Utils::NetStream::ReturnGet c;
|
||||
while ((c = CommsStream.get()).HasData()) {
|
||||
switch (c.GetData()) {
|
||||
case '$': {
|
||||
auto packet = ReadPacket();
|
||||
response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown packet {}", packet);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case '+':
|
||||
// ACK, do nothing.
|
||||
break;
|
||||
case '-':
|
||||
// NAK, Resend requested
|
||||
{
|
||||
std::lock_guard lk(sendMutex);
|
||||
SendPacket(response.Response);
|
||||
}
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
SyscallHandler->TM.Pause();
|
||||
fextl::string str = fextl::fmt::format("T02thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
}
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("GdbServer: Unexpected byte {} ({:02x})", c.GetData(), c.GetData());
|
||||
}
|
||||
}
|
||||
// Receive packet data
|
||||
async_read_until(*CommsSocket, fasio::dynamic_vector_buffer {CommsBuffer}, std::bind_front(&GdbServer::MatchPacket, this),
|
||||
std::bind_front(&GdbServer::HandlePacket, this));
|
||||
|
||||
if (c.HasHangup()) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Repeat to catch disconnect events
|
||||
return fasio::post_callback::repeat;
|
||||
});
|
||||
|
||||
{
|
||||
std::lock_guard lk(sendMutex);
|
||||
CommsStream.InvalidateSocket();
|
||||
}
|
||||
CommsBuffer.reserve(1000);
|
||||
|
||||
// Enter event loop
|
||||
Reactor.run();
|
||||
|
||||
// Shut down
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (CommsSocket) {
|
||||
close(CommsSocket->FD);
|
||||
CommsSocket.reset();
|
||||
}
|
||||
|
||||
CloseListenSocket();
|
||||
@@ -1473,22 +1502,9 @@ void GdbServer::OpenListenSocket() {
|
||||
|
||||
GdbUnixSocketPath = fextl::fmt::format("{}{}-gdb", GdbUnixPath, ::getpid());
|
||||
|
||||
ListenSocket = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (ListenSocket == -1) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
return;
|
||||
}
|
||||
|
||||
struct sockaddr_un addr {};
|
||||
addr.sun_family = AF_UNIX;
|
||||
strncpy(addr.sun_path, GdbUnixSocketPath.data(), sizeof(addr.sun_path));
|
||||
size_t SizeOfAddr = offsetof(sockaddr_un, sun_path) + GdbUnixSocketPath.size();
|
||||
|
||||
// Bind the socket to the path
|
||||
int Result {};
|
||||
for (int attempt = 0; attempt < 2; ++attempt) {
|
||||
Result = bind(ListenSocket, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
|
||||
if (Result == 0) {
|
||||
Acceptor = fasio::tcp_acceptor::create(Reactor, false, GdbUnixSocketPath, 1);
|
||||
if (Acceptor) {
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1497,35 +1513,19 @@ void GdbServer::OpenListenSocket() {
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
|
||||
if (Result != 0) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't bind AF_UNIX socket '{}': {} {}\n", addr.sun_path, errno, strerror(errno));
|
||||
close(ListenSocket);
|
||||
ListenSocket = -1;
|
||||
if (!Acceptor) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't bind AF_UNIX socket '{}': {} {}\n", GdbUnixSocketPath, errno, strerror(errno));
|
||||
return;
|
||||
}
|
||||
|
||||
listen(ListenSocket, 1);
|
||||
LogMan::Msg::IFmt("[GdbServer] Waiting for connection on {}", GdbUnixSocketPath);
|
||||
LogMan::Msg::IFmt("[GdbServer] gdb-multiarch -ex \"set debug remote 1\" -ex \"target extended-remote {}\"", GdbUnixSocketPath);
|
||||
}
|
||||
|
||||
void GdbServer::CloseListenSocket() {
|
||||
if (ListenSocket != -1) {
|
||||
close(ListenSocket);
|
||||
ListenSocket = -1;
|
||||
}
|
||||
Acceptor.reset();
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
|
||||
void GdbServer::OpenSocket() {
|
||||
// Block until a connection arrives
|
||||
struct sockaddr_storage their_addr {};
|
||||
socklen_t addr_size {};
|
||||
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr*)&their_addr, &addr_size);
|
||||
|
||||
CommsStream.OpenSocket(new_fd);
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEX
|
||||
@@ -13,12 +13,12 @@ $end_info$
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <Common/AsyncNet.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "LinuxSyscalls/NetStream.h"
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
namespace FEX {
|
||||
@@ -40,17 +40,11 @@ private:
|
||||
|
||||
void OpenListenSocket();
|
||||
void CloseListenSocket();
|
||||
enum class WaitForConnectionResult {
|
||||
CONNECTION,
|
||||
ERROR,
|
||||
};
|
||||
WaitForConnectionResult WaitForConnection();
|
||||
void OpenSocket();
|
||||
void StartThread();
|
||||
fextl::string ReadPacket();
|
||||
void SendPacket(const fextl::string& packet);
|
||||
fextl::string ReadPacket(const std::span<std::byte>& stream);
|
||||
void SendPacket(fasio::tcp_socket&, const fextl::string& packet);
|
||||
|
||||
void SendACK(bool NACK);
|
||||
void SendACK(fasio::tcp_socket&, bool NACK);
|
||||
|
||||
Event ThreadBreakEvent {};
|
||||
void WaitForThreadWakeup();
|
||||
@@ -147,7 +141,14 @@ private:
|
||||
FEX::HLE::SyscallHandler* const SyscallHandler;
|
||||
FEX::HLE::SignalDelegator* SignalDelegation;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
FEX::Utils::NetStream CommsStream;
|
||||
fasio::poll_reactor Reactor;
|
||||
std::optional<fasio::tcp_acceptor> Acceptor;
|
||||
std::optional<fasio::tcp_socket> CommsSocket;
|
||||
fextl::vector<std::byte> CommsBuffer;
|
||||
|
||||
std::pair<fextl::vector<std::byte>::iterator, bool> MatchPacket(fextl::vector<std::byte>::iterator begin, fextl::vector<std::byte>::iterator end);
|
||||
void HandlePacket(fasio::error ec, size_t BytesInMessage);
|
||||
|
||||
std::mutex sendMutex;
|
||||
bool SettingNoAckMode {false};
|
||||
bool NoAckMode {false};
|
||||
@@ -156,13 +157,11 @@ private:
|
||||
fextl::string OSDataString {};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::atomic<bool> CoreShuttingDown {};
|
||||
fextl::string LibraryMapString {};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, FEX::HLE::SignalDelegator::MAX_SIGNALS + 1> PassSignals {};
|
||||
uint32_t CurrentDebuggingThread {};
|
||||
int ListenSocket {};
|
||||
fextl::string GdbUnixSocketPath {};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
@@ -1,107 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "LinuxSyscalls/NetStream.h"
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstring>
|
||||
#include <sys/eventfd.h>
|
||||
#include <sys/socket.h>
|
||||
#include <poll.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEX::Utils {
|
||||
|
||||
NetStream::NetStream()
|
||||
: receive_buffer(1500) {
|
||||
eventfd = ::eventfd(0, EFD_CLOEXEC);
|
||||
}
|
||||
|
||||
NetStream::ReturnGet NetStream::get() {
|
||||
if (read_offset != receive_buffer.size() && read_offset != receive_offset) {
|
||||
auto Result = receive_buffer.at(read_offset);
|
||||
++read_offset;
|
||||
return NetStream::ReturnGet {Result};
|
||||
}
|
||||
|
||||
if (read_offset == receive_buffer.size()) {
|
||||
read_offset = 0;
|
||||
receive_offset = 0;
|
||||
}
|
||||
|
||||
struct pollfd pfds[2] = {
|
||||
{
|
||||
.fd = socketfd,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
},
|
||||
|
||||
{
|
||||
.fd = eventfd,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
},
|
||||
};
|
||||
|
||||
auto Result = poll(pfds, sizeof(pfds), -1);
|
||||
if (Result > 0) {
|
||||
for (auto& pfd : pfds) {
|
||||
if (pfd.fd == eventfd && (pfd.revents & POLLIN)) {
|
||||
// Interrupted by eventfd.
|
||||
uint64_t data;
|
||||
::read(pfd.fd, &data, sizeof(data));
|
||||
break;
|
||||
}
|
||||
|
||||
if (pfd.revents & POLLHUP) {
|
||||
return NetStream::ReturnGet {true};
|
||||
}
|
||||
|
||||
const auto remaining_size = receive_buffer.size() - receive_offset;
|
||||
Result = ::recv(socketfd, &receive_buffer.at(receive_offset), remaining_size, 0);
|
||||
if (Result > 0) {
|
||||
receive_offset += Result;
|
||||
auto Result = receive_buffer.at(read_offset);
|
||||
++read_offset;
|
||||
return NetStream::ReturnGet {Result};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return NetStream::ReturnGet {false};
|
||||
}
|
||||
|
||||
size_t NetStream::read(char* buf, size_t size, bool ContinueOnInterrupt) {
|
||||
size_t Read {};
|
||||
while (Read < size) {
|
||||
auto Result = get();
|
||||
if (Result.HasData()) {
|
||||
buf[Read] = Result.GetData();
|
||||
++Read;
|
||||
} else if ((!Result.HasData() && !ContinueOnInterrupt) || Result.HasHangup()) {
|
||||
return Read;
|
||||
}
|
||||
}
|
||||
return Read;
|
||||
}
|
||||
|
||||
void NetStream::InterruptReader() {
|
||||
uint64_t data {1};
|
||||
::write(eventfd, &data, sizeof(data));
|
||||
}
|
||||
|
||||
bool NetStream::SendPacket(const fextl::string& packet) {
|
||||
size_t Total {};
|
||||
while (Total < packet.size()) {
|
||||
size_t Remaining = packet.size() - Total;
|
||||
size_t sent = ::send(socketfd, &packet.at(Total), Remaining, MSG_NOSIGNAL);
|
||||
if (sent == -1) {
|
||||
return false;
|
||||
}
|
||||
Total += sent;
|
||||
}
|
||||
|
||||
return Total == packet.size();
|
||||
}
|
||||
|
||||
} // namespace FEX::Utils
|
||||
@@ -1,53 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <optional>
|
||||
#include <variant>
|
||||
|
||||
namespace FEX::Utils {
|
||||
class NetStream final {
|
||||
public:
|
||||
NetStream();
|
||||
|
||||
void OpenSocket(int _socketfd) {
|
||||
socketfd = _socketfd;
|
||||
}
|
||||
|
||||
void InvalidateSocket() {
|
||||
socketfd = -1;
|
||||
}
|
||||
|
||||
bool HasSocket() const {
|
||||
return socketfd != -1;
|
||||
}
|
||||
|
||||
struct ReturnGet final : public std::variant<char, bool> {
|
||||
bool HasHangup() const {
|
||||
return std::holds_alternative<bool>(*this) && std::get<bool>(*this);
|
||||
}
|
||||
bool HasData() const {
|
||||
return std::holds_alternative<char>(*this);
|
||||
}
|
||||
char GetData() const {
|
||||
return std::get<char>(*this);
|
||||
}
|
||||
};
|
||||
|
||||
ReturnGet get();
|
||||
size_t read(char* buf, size_t size, bool ContinueOnInterrupt);
|
||||
|
||||
void InterruptReader();
|
||||
|
||||
bool SendPacket(const fextl::string& packet);
|
||||
private:
|
||||
int socketfd {-1};
|
||||
int eventfd {-1};
|
||||
size_t read_offset {};
|
||||
size_t receive_offset {};
|
||||
fextl::vector<char> receive_buffer;
|
||||
};
|
||||
} // namespace FEX::Utils
|
||||
@@ -257,7 +257,9 @@ std::optional<int> SeccompEmulator::SerializeFilters(FEXCore::Core::CpuStateFram
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
|
||||
// Seal everything about this FD.
|
||||
fcntl(FD, F_ADD_SEALS, F_SEAL_SEAL | F_SEAL_SHRINK | F_SEAL_GROW | F_SEAL_WRITE | F_SEAL_FUTURE_WRITE);
|
||||
if (fcntl(FD, F_ADD_SEALS, F_SEAL_SEAL | F_SEAL_SHRINK | F_SEAL_GROW | F_SEAL_WRITE | F_SEAL_FUTURE_WRITE) == -1) {
|
||||
LogMan::Msg::IFmt("Couldn't seal seccomp serialize FD. Nefarious code could modify");
|
||||
}
|
||||
|
||||
return FD;
|
||||
}
|
||||
@@ -410,7 +412,7 @@ SeccompEmulator::ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JIT
|
||||
case SECCOMP_RET_KILL_PROCESS: {
|
||||
const int KillSignal = GetKillSignal();
|
||||
// Ignores signal handler and sigmask
|
||||
uint64_t Mask = 1 << (KillSignal - 1);
|
||||
uint64_t Mask = 1ULL << (KillSignal - 1);
|
||||
SignalDelegation->GuestSigProcMask(Thread, SIG_UNBLOCK, &Mask, nullptr);
|
||||
SignalDelegation->UninstallHostHandler(KillSignal);
|
||||
kill(0, KillSignal);
|
||||
|
||||
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
@@ -59,6 +60,7 @@ static FEX::HLE::ThreadStateObject* GetThreadFromAltStack(const stack_t& alt_sta
|
||||
static void SignalHandlerThunk(int Signal, siginfo_t* Info, void* UContext) {
|
||||
ucontext_t* _context = (ucontext_t*)UContext;
|
||||
auto ThreadObject = GetThreadFromAltStack(_context->uc_stack);
|
||||
FEXCORE_PROFILE_ACCUMULATION(ThreadObject->Thread, AccumulatedSignalTime);
|
||||
ThreadObject->SignalInfo.Delegator->HandleSignal(ThreadObject, Signal, Info, UContext);
|
||||
}
|
||||
|
||||
@@ -673,6 +675,8 @@ void SignalDelegator::HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObjec
|
||||
SaveTelemetry();
|
||||
#endif
|
||||
|
||||
FEX::HLE::_SyscallHandler->TM.CleanupForExit();
|
||||
|
||||
// Reassign back to DFL and crash
|
||||
signal(Signal, SIG_DFL);
|
||||
if (SigInfo.si_code != SI_KERNEL) {
|
||||
@@ -916,6 +920,7 @@ SignalDelegator::SignalDelegator(FEXCore::Context::Context* _CTX, const std::str
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedSIGBUSCount, 1);
|
||||
const auto Delegator = FEX::HLE::ThreadManager::GetStateObjectFromFEXCoreThread(Thread)->SignalInfo.Delegator;
|
||||
const auto Result = FEXCore::ArchHelpers::Arm64::HandleUnalignedAccess(Thread, Delegator->GetUnalignedHandlerType(), PC,
|
||||
ArchHelpers::Context::GetArmGPRs(ucontext));
|
||||
|
||||
@@ -883,6 +883,7 @@ uint64_t UnimplementedSyscallSafe(FEXCore::Core::CpuStateFrame* Frame, uint64_t
|
||||
}
|
||||
|
||||
void SyscallHandler::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
TM.LockBeforeFork();
|
||||
Thread->CTX->LockBeforeFork(Thread);
|
||||
VMATracking.Mutex.lock();
|
||||
}
|
||||
@@ -937,7 +938,7 @@ SyscallHandler::GenerateMap(const std::string_view& GuestBinaryFile, const std::
|
||||
return {};
|
||||
}
|
||||
|
||||
const auto GuestSourceFile = fextl::fmt::format("{}/{}.src", FexSrcPath, GuestBinaryFileId);
|
||||
auto GuestSourceFile = fextl::fmt::format("{}/{}.src", FexSrcPath, GuestBinaryFileId);
|
||||
|
||||
struct stat GuestSourceFileStat;
|
||||
|
||||
@@ -1060,7 +1061,7 @@ DoGenerate:
|
||||
|
||||
auto rv = fextl::make_unique<FEXCore::HLE::SourcecodeMap>();
|
||||
|
||||
rv->SourceFile = GuestSourceFile;
|
||||
rv->SourceFile = std::move(GuestSourceFile);
|
||||
|
||||
auto EndSymbol = [&] {
|
||||
if (LastSymbolOffset) {
|
||||
|
||||
@@ -61,6 +61,9 @@ static void* ThreadHandler(void* Data) {
|
||||
|
||||
Thread->ThreadInfo.PID = ::getpid();
|
||||
Thread->ThreadInfo.TID = FHU::Syscalls::gettid();
|
||||
if (Thread->Thread->ThreadStats) {
|
||||
Thread->Thread->ThreadStats->TID.store(Thread->ThreadInfo.TID, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
FEX::HLE::_SyscallHandler->RegisterTLSState(Thread);
|
||||
|
||||
@@ -558,6 +561,7 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int status) -> uint64_t {
|
||||
// Save telemetry if we're exiting.
|
||||
FEX::HLE::_SyscallHandler->GetSignalDelegator()->SaveTelemetry();
|
||||
FEX::HLE::_SyscallHandler->TM.CleanupForExit();
|
||||
|
||||
syscall(SYSCALL_DEF(exit_group), status);
|
||||
// This will never be reached
|
||||
|
||||
@@ -97,6 +97,7 @@ bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState* Thread,
|
||||
});
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedSMCCount, 1);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,8 +4,155 @@
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <fcntl.h>
|
||||
#include <git_version.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
|
||||
ThreadManager::StatAlloc::StatAlloc() {
|
||||
Initialize();
|
||||
SaveHeader(Is64BitMode() ? FEXCore::Profiler::AppType::LINUX_64 : FEXCore::Profiler::AppType::LINUX_32);
|
||||
}
|
||||
|
||||
void ThreadManager::StatAlloc::Initialize() {
|
||||
if (!ProfileStats()) {
|
||||
return;
|
||||
}
|
||||
|
||||
int fd = shm_open(fextl::fmt::format("fex-{}-stats", ::getpid()).c_str(), O_CREAT | O_TRUNC | O_RDWR, USER_PERMS);
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
CurrentSize = sysconf(_SC_PAGESIZE);
|
||||
CurrentSize = CurrentSize > 0 ? CurrentSize : FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
if (ftruncate(fd, CurrentSize) == -1) {
|
||||
LogMan::Msg::EFmt("[StatAlloc] ftruncate failed");
|
||||
goto err;
|
||||
}
|
||||
|
||||
// Reserve a region of MAX_STATS_SIZE so we can grow the allocation buffer.
|
||||
// Number of thread slots when ThreadStatsHeader == 64bytes and ThreadStats == 40bytes:
|
||||
// 1 page: 99 slots
|
||||
// 1 MB: 26211 slots
|
||||
// 128 MB: 3355440 slots
|
||||
Base = ::mmap(nullptr, MAX_STATS_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0);
|
||||
if (Base == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("[StatAlloc] mmap base failed");
|
||||
Base = nullptr;
|
||||
goto err;
|
||||
}
|
||||
|
||||
// Allocate a small working shared space for now, grow as necessary.
|
||||
{
|
||||
auto SharedBase = ::mmap(Base, CurrentSize, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_FIXED, fd, 0);
|
||||
if (SharedBase == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("[StatAlloc] mmap shm failed");
|
||||
munmap(Base, MAX_STATS_SIZE);
|
||||
Base = nullptr;
|
||||
goto err;
|
||||
}
|
||||
}
|
||||
|
||||
err:
|
||||
close(fd);
|
||||
}
|
||||
|
||||
uint32_t ThreadManager::StatAlloc::FrontendAllocateSlots(uint32_t NewSize) {
|
||||
if (CurrentSize == MAX_STATS_SIZE) {
|
||||
// Allocator has reached maximum slots. We can't allocate anymore.
|
||||
// New threads won't get stats.
|
||||
return CurrentSize;
|
||||
}
|
||||
NewSize = std::max(MAX_STATS_SIZE, NewSize);
|
||||
|
||||
// When allocating more slots, open the fd without O_TRUNC | O_CREAT.
|
||||
int fd = shm_open(fextl::fmt::format("fex-{}-stats", ::getpid()).c_str(), O_RDWR, USER_PERMS);
|
||||
if (!fd) {
|
||||
return CurrentSize;
|
||||
}
|
||||
|
||||
if (ftruncate(fd, NewSize) == -1) {
|
||||
LogMan::Msg::EFmt("[StatAlloc] ftruncate more failed");
|
||||
|
||||
goto err;
|
||||
}
|
||||
|
||||
{
|
||||
auto SharedBase = ::mmap(Base, NewSize, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_FIXED, fd, 0);
|
||||
if (SharedBase == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("[StatAlloc] allocate more mmap shm failed");
|
||||
goto err;
|
||||
}
|
||||
}
|
||||
|
||||
err:
|
||||
close(fd);
|
||||
return NewSize;
|
||||
}
|
||||
|
||||
FEXCore::Profiler::ThreadStats* ThreadManager::StatAlloc::AllocateSlot(uint32_t TID) {
|
||||
std::scoped_lock lk(StatMutex);
|
||||
return StatAllocBase::AllocateSlot(TID);
|
||||
}
|
||||
|
||||
void ThreadManager::StatAlloc::DeallocateSlot(FEXCore::Profiler::ThreadStats* AllocatedSlot) {
|
||||
if (!AllocatedSlot) {
|
||||
return;
|
||||
}
|
||||
|
||||
std::scoped_lock lk(StatMutex);
|
||||
StatAllocBase::DeallocateSlot(AllocatedSlot);
|
||||
}
|
||||
|
||||
void ThreadManager::StatAlloc::CleanupForExit() {
|
||||
shm_unlink(fextl::fmt::format("fex-{}-stats", ::getpid()).c_str());
|
||||
}
|
||||
|
||||
void ThreadManager::StatAlloc::LockBeforeFork() {
|
||||
if (!ProfileStats()) {
|
||||
return;
|
||||
}
|
||||
StatMutex.lock();
|
||||
}
|
||||
|
||||
void ThreadManager::StatAlloc::UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child) {
|
||||
if (!ProfileStats()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!Child) {
|
||||
StatMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
StatMutex.StealAndDropActiveLocks();
|
||||
|
||||
// shm_memory ownership is retained by the parent process, so the child must replace it with its own one.
|
||||
// Otherwise this process will keep reporting in the original parent thread's stats region.
|
||||
munmap(Base, MAX_STATS_SIZE);
|
||||
Base = nullptr;
|
||||
CurrentSize = 0;
|
||||
Head = nullptr;
|
||||
Stats = nullptr;
|
||||
StatTail = nullptr;
|
||||
RemainingSlots = 0;
|
||||
|
||||
Thread->ThreadStats = nullptr;
|
||||
|
||||
Initialize();
|
||||
SaveHeader(Is64BitMode() ? FEXCore::Profiler::AppType::LINUX_64 : FEXCore::Profiler::AppType::LINUX_32);
|
||||
|
||||
// Update this thread's ThreadStats object
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromFEXCoreThread(Thread);
|
||||
ThreadObject->Thread->ThreadStats = AllocateSlot(ThreadObject->ThreadInfo.TID);
|
||||
}
|
||||
|
||||
FEX::HLE::ThreadStateObject* ThreadManager::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState,
|
||||
uint64_t ParentTID, FEX::HLE::ThreadStateObject* InheritThread) {
|
||||
auto ThreadStateObject = new FEX::HLE::ThreadStateObject;
|
||||
@@ -13,12 +160,13 @@ FEX::HLE::ThreadStateObject* ThreadManager::CreateThread(uint64_t InitialRIP, ui
|
||||
ThreadStateObject->ThreadInfo.parent_tid = ParentTID;
|
||||
ThreadStateObject->ThreadInfo.PID = ::getpid();
|
||||
|
||||
if (ParentTID == 0) {
|
||||
ThreadStateObject->ThreadInfo.TID = FHU::Syscalls::gettid();
|
||||
}
|
||||
ThreadStateObject->ThreadInfo.TID = FHU::Syscalls::gettid();
|
||||
|
||||
ThreadStateObject->Thread = CTX->CreateThread(InitialRIP, StackPointer, NewThreadState, ParentTID);
|
||||
ThreadStateObject->Thread->FrontendPtr = ThreadStateObject;
|
||||
if (ProfileStats()) {
|
||||
ThreadStateObject->Thread->ThreadStats = Stat.AllocateSlot(ThreadStateObject->ThreadInfo.TID);
|
||||
}
|
||||
|
||||
if (InheritThread) {
|
||||
FEX::HLE::_SyscallHandler->SeccompEmulator.InheritSeccompFilters(InheritThread, ThreadStateObject);
|
||||
@@ -37,6 +185,8 @@ void ThreadManager::DestroyThread(FEX::HLE::ThreadStateObject* Thread, bool Need
|
||||
Threads.erase(It);
|
||||
}
|
||||
|
||||
Stat.DeallocateSlot(Thread->Thread->ThreadStats);
|
||||
|
||||
HandleThreadDeletion(Thread, NeedsTLSUninstall);
|
||||
}
|
||||
|
||||
@@ -212,7 +362,12 @@ void ThreadManager::UnpauseThread(FEX::HLE::ThreadStateObject* Thread) {
|
||||
Thread->ThreadPaused.NotifyOne();
|
||||
}
|
||||
|
||||
void ThreadManager::LockBeforeFork() {
|
||||
Stat.LockBeforeFork();
|
||||
}
|
||||
|
||||
void ThreadManager::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread, bool Child) {
|
||||
Stat.UnlockAfterFork(LiveThread, Child);
|
||||
if (!Child) {
|
||||
return;
|
||||
}
|
||||
@@ -220,6 +375,9 @@ void ThreadManager::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThre
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto& DeadThread : Threads) {
|
||||
// The fork parent retains ownership of ThreadStats
|
||||
DeadThread->Thread->ThreadStats = nullptr;
|
||||
|
||||
if (DeadThread->Thread == LiveThread) {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -8,11 +8,14 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "Common/Profiler.h"
|
||||
|
||||
#include "LinuxSyscalls/Types.h"
|
||||
#include "LinuxSyscalls/Seccomp/SeccompEmulator.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
#include <cstdint>
|
||||
@@ -105,6 +108,35 @@ public:
|
||||
|
||||
~ThreadManager();
|
||||
|
||||
class StatAlloc final : public FEX::Profiler::StatAllocBase {
|
||||
public:
|
||||
StatAlloc();
|
||||
|
||||
void LockBeforeFork();
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child);
|
||||
|
||||
void CleanupForExit();
|
||||
|
||||
FEXCore::Profiler::ThreadStats* AllocateSlot(uint32_t TID);
|
||||
void DeallocateSlot(FEXCore::Profiler::ThreadStats* AllocatedSlot);
|
||||
|
||||
private:
|
||||
void Initialize();
|
||||
|
||||
uint32_t FrontendAllocateSlots(uint32_t NewSize) override;
|
||||
FEX_CONFIG_OPT(ProfileStats, PROFILESTATS);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
constexpr static int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
FEXCore::ForkableUniqueMutex StatMutex;
|
||||
};
|
||||
|
||||
void CleanupForExit() {
|
||||
Stat.CleanupForExit();
|
||||
}
|
||||
|
||||
StatAlloc Stat;
|
||||
|
||||
///< Returns the ThreadStateObject from a CpuStateFrame object.
|
||||
static inline FEX::HLE::ThreadStateObject* GetStateObjectFromCPUState(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
return static_cast<FEX::HLE::ThreadStateObject*>(Frame->Thread->FrontendPtr);
|
||||
@@ -136,6 +168,7 @@ public:
|
||||
|
||||
void SleepThread(FEXCore::Context::Context* CTX, FEXCore::Core::CpuStateFrame* Frame);
|
||||
|
||||
void LockBeforeFork();
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child);
|
||||
|
||||
void IncrementIdleRefCount() {
|
||||
@@ -188,6 +221,7 @@ private:
|
||||
|
||||
void HandleThreadDeletion(FEX::HLE::ThreadStateObject* Thread, bool NeedsTLSUninstall = false);
|
||||
void NotifyPause();
|
||||
FEX_CONFIG_OPT(ProfileStats, PROFILESTATS);
|
||||
};
|
||||
|
||||
} // namespace FEX::HLE
|
||||
@@ -652,7 +652,7 @@ void LoadGuestVDSOSymbols(bool Is64Bit, char* VDSOBase) {
|
||||
void LoadUnique32BitSigreturn(VDSOMapping* Mapping) {
|
||||
// Hardcoded to one page for now
|
||||
const auto PageSize = sysconf(_SC_PAGESIZE);
|
||||
Mapping->OptionalMappingSize = PageSize;
|
||||
Mapping->OptionalMappingSize = PageSize > 0 ? PageSize : FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
// First 64bit page
|
||||
constexpr uintptr_t LOCATION_MAX = 0x1'0000'0000;
|
||||
|
||||
+120
-11
@@ -1,6 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "OptionParser.h"
|
||||
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <charconv>
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
@@ -63,6 +65,43 @@ void LoadOptions(int argc, char** argv) {
|
||||
}
|
||||
} // namespace Config
|
||||
|
||||
bool FindWineFEXApplication(int64_t PID, std::string_view exe, const std::vector<std::string_view>& Args) {
|
||||
// Walk the arguments and see if anything contains wine.
|
||||
bool FoundWine = false;
|
||||
|
||||
if (exe.find("wine") != exe.npos) {
|
||||
FoundWine = true;
|
||||
}
|
||||
|
||||
if (!FoundWine) {
|
||||
for (auto Arg : Args) {
|
||||
if (Arg.find("wine") != Arg.npos) {
|
||||
FoundWine = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!FoundWine) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Wine was found, scan the mapped files to see if anything mapped "libarm64ecfex.dll" or "libwow64fex.dll"
|
||||
for (const auto& Entry : std::filesystem::directory_iterator(fmt::format("/proc/{}/map_files", PID))) {
|
||||
// If not a symlink then skip.
|
||||
if (!Entry.is_symlink()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto filename = std::filesystem::read_symlink(Entry.path()).filename().string();
|
||||
if (filename.find("arm64ecfex.dll") != filename.npos || filename.find("wow64fex.dll") != filename.npos) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
struct PIDInfo {
|
||||
int64_t pid;
|
||||
std::string cmdline;
|
||||
@@ -155,7 +194,7 @@ int main(int argc, char** argv) {
|
||||
PIDs.emplace_back(PIDInfo {
|
||||
.pid = pid,
|
||||
.cmdline = CMDLineData.str(),
|
||||
.exe_link = exe_link,
|
||||
.exe_link = std::move(exe_link),
|
||||
.State = State,
|
||||
});
|
||||
}
|
||||
@@ -177,28 +216,98 @@ int main(int argc, char** argv) {
|
||||
arg += strlen(arg) + 1;
|
||||
}
|
||||
|
||||
auto IsFEX = [](auto& Path) {
|
||||
auto FindFEXArgument = [](auto& Path) -> int32_t {
|
||||
if (Path.ends_with("FEXInterpreter")) {
|
||||
return true;
|
||||
return 1;
|
||||
}
|
||||
if (Path.ends_with("FEXLoader")) {
|
||||
return true;
|
||||
return 1;
|
||||
}
|
||||
|
||||
return false;
|
||||
return -1;
|
||||
};
|
||||
bool IsFEXBin = IsFEX(pid.exe_link) || IsFEX(Args[0]);
|
||||
if (!IsFEXBin) {
|
||||
|
||||
struct ProgramPair {
|
||||
std::string_view ProgramPath;
|
||||
std::string_view ProgramFilename;
|
||||
};
|
||||
|
||||
auto FindEmulatedWineArgument = [](int32_t BeginningArg, const std::vector<std::string_view>& Args, bool Wine) -> ProgramPair {
|
||||
std::string_view ProgramName = Args[BeginningArg];
|
||||
|
||||
for (size_t i = BeginningArg; i < Args.size(); ++i) {
|
||||
auto CurrentProgramName = FHU::Filesystem::GetFilename(Args[i]);
|
||||
|
||||
if (CurrentProgramName == "wine-preloader" || CurrentProgramName == "wine64-preloader") {
|
||||
// Wine preloader is required to be in the format of `wine-preloader <wine executable>`
|
||||
// The preloader doesn't execve the executable, instead maps it directly itself
|
||||
// Skip the next argument since we know it is wine (potentially with custom wine executable name)
|
||||
++i;
|
||||
Wine = true;
|
||||
} else if (CurrentProgramName == "wine" || CurrentProgramName == "wine64") {
|
||||
// Next argument, this isn't the program we want
|
||||
//
|
||||
// If we are running wine or wine64 then we should check the next argument for the application name instead.
|
||||
// wine will change the active program name with `setprogname` or `prctl(PR_SET_NAME`.
|
||||
// Since FEX needs this data far earlier than libraries we need a different check.
|
||||
Wine = true;
|
||||
} else {
|
||||
if (Wine == true) {
|
||||
// If this was path separated with '\' then we need to check that.
|
||||
auto WinSeparator = CurrentProgramName.find_last_of('\\');
|
||||
if (WinSeparator != CurrentProgramName.npos) {
|
||||
// Used windows separators
|
||||
CurrentProgramName = CurrentProgramName.substr(WinSeparator + 1);
|
||||
}
|
||||
|
||||
return {
|
||||
.ProgramPath = Args[i],
|
||||
.ProgramFilename = CurrentProgramName,
|
||||
};
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
auto ProgramFilename = ProgramName;
|
||||
auto Separator = ProgramName.find_last_of('/');
|
||||
if (Separator != ProgramName.npos) {
|
||||
// Used windows separators
|
||||
ProgramFilename = ProgramFilename.substr(Separator + 1);
|
||||
}
|
||||
|
||||
return {
|
||||
.ProgramPath = ProgramName,
|
||||
.ProgramFilename = ProgramFilename,
|
||||
};
|
||||
};
|
||||
|
||||
int32_t ProgramArg = -1;
|
||||
ProgramArg = FindFEXArgument(pid.exe_link);
|
||||
if (ProgramArg == -1) {
|
||||
ProgramArg = FindFEXArgument(Args[0]);
|
||||
}
|
||||
|
||||
bool IsWine = false;
|
||||
if (ProgramArg == -1) {
|
||||
// If we still haven't found a FEXInterpreter path then this might be an arm64ec FEX application.
|
||||
// The only way to know for sure is the walk the mapped files of the process and check if FEX is mapped.
|
||||
if (FindWineFEXApplication(pid.pid, pid.exe_link, Args)) {
|
||||
// Search from the start.
|
||||
ProgramArg = 0;
|
||||
IsWine = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (ProgramArg == -1 || ProgramArg >= Args.size()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto Arg1 = Args[1];
|
||||
auto Arg1Program = std::filesystem::path(Arg1).filename();
|
||||
|
||||
ProgramPair Arg = FindEmulatedWineArgument(ProgramArg, Args, IsWine);
|
||||
bool Matched = false;
|
||||
for (const auto& CompareProgram : Config::Programs) {
|
||||
auto CompareProgramFilename = std::filesystem::path(CompareProgram).filename();
|
||||
if (CompareProgram == Arg1Program || CompareProgram == Arg1 || CompareProgramFilename == Arg1Program) {
|
||||
if (CompareProgram == Arg.ProgramFilename || CompareProgram == Arg.ProgramPath || CompareProgramFilename == Arg.ProgramFilename) {
|
||||
MatchedPIDs.emplace(pid.pid);
|
||||
Matched = true;
|
||||
break;
|
||||
|
||||
Loaded 100 of 172 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user