mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 15:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cae4f2f873 | ||
|
|
0fc6d6b6b5 | ||
|
|
7227ee9b2e | ||
|
|
c6153d6a52 | ||
|
|
d82d2944a9 | ||
|
|
58ad400519 | ||
|
|
d23c76d0a9 | ||
|
|
6a5b9e2a93 | ||
|
|
f33a93b0a1 | ||
|
|
015200f511 | ||
|
|
92f48819b6 | ||
|
|
0c6483cad5 | ||
|
|
ee02b1ca51 | ||
|
|
33845a3112 | ||
|
|
096ed29b5e | ||
|
|
3bbff8a948 | ||
|
|
726918b82c | ||
|
|
0f59a18223 | ||
|
|
3402cde334 | ||
|
|
8f53c6bb96 | ||
|
|
0d6e4631a3 | ||
|
|
8dd9a5bd38 | ||
|
|
903cf84874 | ||
|
|
e997da48c7 | ||
|
|
5ff89fd171 | ||
|
|
fad4254c0e | ||
|
|
2fb3c4f11c | ||
|
|
ce5297b75f | ||
|
|
b2b0c277f6 | ||
|
|
46919979ce | ||
|
|
8b716c6a22 | ||
|
|
75090f8f6c | ||
|
|
c14c0c2e3b | ||
|
|
95efd18b73 | ||
|
|
c9319a768f | ||
|
|
da48020882 | ||
|
|
ce4380e136 | ||
|
|
c633661121 | ||
|
|
4bfd1dde1f | ||
|
|
fe11bd2242 | ||
|
|
814f0c3c93 | ||
|
|
7e904056d3 | ||
|
|
969d8f866c | ||
|
|
a2d0b7d7c4 | ||
|
|
97a8fa77bc | ||
|
|
c1296cc64d | ||
|
|
1dee54a9d8 | ||
|
|
17e5d73e64 | ||
|
|
d523b7a6c7 | ||
|
|
24ad208778 | ||
|
|
fa87c73b9e | ||
|
|
0ed96544e1 | ||
|
|
d28ccc59ac | ||
|
|
eaa75c1ed2 | ||
|
|
4f4263263b | ||
|
|
ae00654694 | ||
|
|
a7156276e9 | ||
|
|
29859d2491 | ||
|
|
5460a24ea9 | ||
|
|
a284adcd19 | ||
|
|
73d43c1d55 | ||
|
|
256df76674 | ||
|
|
c8dc663b0b | ||
|
|
ba78dff1f8 | ||
|
|
1e597bfbed | ||
|
|
b78af2fdaf | ||
|
|
5379f0a9c7 | ||
|
|
f1f523e525 | ||
|
|
c79d79e08b | ||
|
|
ee2d417d21 | ||
|
|
cebdde599a | ||
|
|
c3ac72a01e | ||
|
|
13f3c6e75a | ||
|
|
b3cd4edb3b | ||
|
|
afe10c1666 | ||
|
|
d9d30916ba | ||
|
|
3f6c1c0e68 | ||
|
|
5b2cc77109 | ||
|
|
b824023ec6 | ||
|
|
6ce1be0880 | ||
|
|
c5dacab2ee | ||
|
|
4d24b85d57 | ||
|
|
e967b447e6 | ||
|
|
9bc631a427 | ||
|
|
c480ef137d | ||
|
|
15629e790e | ||
|
|
3f08d8b691 | ||
|
|
9d9d171aad | ||
|
|
65218c8285 | ||
|
|
27f2e0b06d | ||
|
|
ae75983b54 | ||
|
|
f8ba373e18 | ||
|
|
2feae06209 | ||
|
|
2e0534924a | ||
|
|
ad1fd7f54b | ||
|
|
9aaace51e1 | ||
|
|
58841142ee | ||
|
|
a6a816fb38 | ||
|
|
70988ccfee | ||
|
|
a8d9caf0c0 | ||
|
|
bc22186093 | ||
|
|
317416b2e0 | ||
|
|
b5ae9e4c97 | ||
|
|
099737ca05 | ||
|
|
e90164b519 | ||
|
|
933c1af7e8 | ||
|
|
b9d878b1f4 | ||
|
|
45a9a83c79 | ||
|
|
560cfc757c | ||
|
|
e92f51e415 | ||
|
|
278ca52d97 | ||
|
|
912dbfe5bd | ||
|
|
687f46fc71 | ||
|
|
6e9e5b3bd6 | ||
|
|
4fbc266b18 | ||
|
|
d1ac406895 | ||
|
|
ce0f5db6f7 | ||
|
|
efb42c1ad1 | ||
|
|
d8109880f4 | ||
|
|
09be28a443 | ||
|
|
fb0bb8dd2c | ||
|
|
8e36f5331f | ||
|
|
a365a70275 | ||
|
|
b5a4e5920d | ||
|
|
94d2ed85a7 | ||
|
|
90f338d7db | ||
|
|
df78f5d50e | ||
|
|
b2b4c2bdcf | ||
|
|
a888da436b | ||
|
|
3cb8ae9a9c | ||
|
|
db3854e391 | ||
|
|
05b4b095fe | ||
|
|
93926641d9 | ||
|
|
89d6752d3d | ||
|
|
e4f95fec79 | ||
|
|
8a7f39559c | ||
|
|
753d0ede6c | ||
|
|
da2e44d024 | ||
|
|
7b379fc3cf | ||
|
|
ec38d58b37 | ||
|
|
f6a74a710d | ||
|
|
3fd136b0da | ||
|
|
72e82d0304 | ||
|
|
2f7dcb8d93 | ||
|
|
128a24d699 | ||
|
|
cd94a8f0ac | ||
|
|
df5e0e5df9 | ||
|
|
458bbf4ef7 | ||
|
|
82319c9deb | ||
|
|
3bc4df7295 | ||
|
|
42a6320935 | ||
|
|
843fe378db | ||
|
|
3679673d5b | ||
|
|
3a269f04d2 | ||
|
|
0021723b50 | ||
|
|
1c4b0272e8 | ||
|
|
9640216124 | ||
|
|
0f8f2bf2b4 | ||
|
|
ef899b7b1a | ||
|
|
b85abf725d | ||
|
|
3f89e46d66 | ||
|
|
d02712ddc6 | ||
|
|
91f48c63ff | ||
|
|
cd16769e57 | ||
|
|
0a778e802f | ||
|
|
d4d5f4d1dd | ||
|
|
b1033ed7c6 | ||
|
|
253333a4cf | ||
|
|
50595ac3a9 |
No files matched your search
+10
-4
@@ -7,7 +7,8 @@ option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with LLD" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
@@ -100,9 +101,14 @@ if (ENABLE_COMPILE_TIME_TRACE)
|
||||
endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
if (ENABLE_LLD)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
link_libraries(${LD_OVERRIDE})
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
@@ -125,7 +131,7 @@ endif()
|
||||
|
||||
if (ENABLE_STATIC_PIE)
|
||||
if (_M_ARM_64 AND ENABLE_LLD)
|
||||
message (FATAL_ERROR "Static linking does not currently work with AArch64+LLD. Use GNU ld for now.")
|
||||
message (FATAL_ERROR "Static linking does not currently work with AArch64+lld. Use GNU ld for now.")
|
||||
endif()
|
||||
|
||||
file(WRITE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"AdditionalArguments": "--no-sandbox"
|
||||
}
|
||||
}
|
||||
+8
-2
@@ -80,16 +80,19 @@ set (SRCS
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/CompileService.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
Interface/Core/OpcodeDispatcher/Vector.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
@@ -184,7 +187,9 @@ if (ENABLE_JIT_X86_64)
|
||||
Interface/Core/JIT/x86_64/MemoryOps.cpp
|
||||
Interface/Core/JIT/x86_64/MiscOps.cpp
|
||||
Interface/Core/JIT/x86_64/MoveOps.cpp
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp)
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp
|
||||
Interface/Core/JIT/x86_64/x64Relocations.cpp
|
||||
)
|
||||
list(APPEND DEFINES -DJIT_X86_64)
|
||||
endif()
|
||||
|
||||
@@ -329,6 +334,7 @@ function(AddDefaultOptionsToTarget Name)
|
||||
|
||||
-Wno-trigraphs
|
||||
-ffunction-sections
|
||||
-fwrapv
|
||||
)
|
||||
|
||||
if (GCC_COLOR)
|
||||
|
||||
@@ -34,6 +34,14 @@ namespace FEXCore {
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (!fp) return;
|
||||
|
||||
|
||||
+1
@@ -13,6 +13,7 @@ public:
|
||||
|
||||
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
|
||||
|
||||
@@ -423,6 +423,16 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
// If running the interpreter then disable cache code compilation
|
||||
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
|
||||
+30
-4
@@ -38,6 +38,17 @@
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
"TextDefault": "none",
|
||||
"Choices": [ "none", "read", "readwrite" ],
|
||||
"ArgumentHandler": "CacheObjectCodeHandler",
|
||||
"Desc": [
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
@@ -104,6 +115,13 @@
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
|
||||
]
|
||||
},
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
@@ -223,14 +241,15 @@
|
||||
"Hacks": {
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MMAN",
|
||||
"TextDefault": "mman",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
"Checks code for modification before execution.",
|
||||
"\tnone: No checks",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap",
|
||||
"\tfull: Validate code before every run (slow)"
|
||||
"\tmtrack: Page tracking based invalidation",
|
||||
"\tfull: Validate code before every run (slow)",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
|
||||
]
|
||||
},
|
||||
"TSOEnabled": {
|
||||
@@ -241,6 +260,13 @@
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
+5
-5
@@ -149,7 +149,7 @@ namespace FEXCore::Context {
|
||||
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->CleanupAfterFork(Thread);
|
||||
}
|
||||
|
||||
|
||||
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
|
||||
CTX->SignalDelegation = SignalDelegation;
|
||||
}
|
||||
@@ -186,11 +186,11 @@ namespace FEXCore::Context {
|
||||
CTX->WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
|
||||
return CTX->AddNamedRegion(Base, Length, Offset, Name);
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
|
||||
return CTX->LoadAOTIRCacheEntry(Name);
|
||||
}
|
||||
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
|
||||
return CTX->RemoveNamedRegion(Base, Length);
|
||||
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
|
||||
return CTX->UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
|
||||
+27
-17
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -34,6 +35,10 @@ class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
}
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
@@ -98,6 +103,8 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
@@ -155,11 +162,13 @@ namespace FEXCore::Context {
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
// Must be called from owning thread
|
||||
static void RemoveThreadCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState
|
||||
static void RemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
RemoveCodeEntry(Frame->Thread, GuestRIP);
|
||||
// Must be called from owning thread
|
||||
static void RemoveThreadCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
RemoveThreadCodeEntry(Frame->Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// Debugger interface
|
||||
@@ -196,18 +205,6 @@ namespace FEXCore::Context {
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
* @param CompileThread Is this for the compile service or not?
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
* This is exposed because the CompileService needs to initialize compilers while copying data from
|
||||
* the paired InternalThreadState that it is compiling code for
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread
|
||||
@@ -269,8 +266,8 @@ namespace FEXCore::Context {
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
|
||||
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
@@ -297,6 +294,9 @@ namespace FEXCore::Context {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
|
||||
@@ -310,6 +310,15 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
@@ -323,6 +332,7 @@ namespace FEXCore::Context {
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
+155
-108
@@ -513,7 +513,8 @@ uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
//Only 32-bit pairs
|
||||
for(int i = 1; i < 10; i++) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg1 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::CCMP_MASK) == FEXCore::ArchHelpers::Arm64::CCMP_INST) {
|
||||
ExpectedReg2 = GetRmReg(NextInstr);
|
||||
@@ -1579,7 +1580,7 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
@@ -1592,7 +1593,7 @@ bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
if (Size == 2) {
|
||||
auto Res = DoLoad16(Addr);
|
||||
@@ -1622,7 +1623,7 @@ bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
@@ -1635,7 +1636,7 @@ bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
constexpr bool DoRetry = false;
|
||||
if (Size == 2) {
|
||||
@@ -1745,7 +1746,8 @@ static uint64_t HandleCAS_NoAtomics(void *_ucontext, void *_info)
|
||||
#endif
|
||||
DesiredReg = GetRdReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg = GetRmReg(NextInstr);
|
||||
}
|
||||
}
|
||||
@@ -1828,11 +1830,13 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
// Scan forward at most five instructions to find our instructions
|
||||
for (size_t i = 1; i < 6; ++i) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_SHIFT_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ADD;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_SHIFT_INST) {
|
||||
uint32_t RnReg = GetRnReg(NextInstr);
|
||||
if (RnReg == REGISTER_MASK) {
|
||||
// Zero reg means neg
|
||||
@@ -1843,21 +1847,34 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST ) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::AND_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_AND;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::BIC_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_BIC;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::OR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_OR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ORN_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ORN;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EOR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EOR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EON_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EON;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
@@ -1888,40 +1905,53 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
constexpr bool DoRetry = true;
|
||||
|
||||
auto NOPExpected = []<typename AtomicType>(AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto BICDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & ~Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto ORNDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | ~Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto EONDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ ~Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = []<typename AtomicType>(AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
if (Size == 2) {
|
||||
using AtomicType = uint16_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -1937,12 +1967,21 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
@@ -1966,38 +2005,6 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
else if (Size == 4) {
|
||||
using AtomicType = uint32_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -2013,12 +2020,21 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
@@ -2042,38 +2058,6 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
}
|
||||
else if (Size == 8) {
|
||||
using AtomicType = uint64_t;
|
||||
auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType {
|
||||
return SrcVal;
|
||||
};
|
||||
|
||||
auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal + Desired;
|
||||
};
|
||||
|
||||
auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal - Desired;
|
||||
};
|
||||
|
||||
auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal & Desired;
|
||||
};
|
||||
|
||||
auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal | Desired;
|
||||
};
|
||||
|
||||
auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return SrcVal ^ Desired;
|
||||
};
|
||||
|
||||
auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return -SrcVal;
|
||||
};
|
||||
|
||||
auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
CASDesiredFn<AtomicType> DesiredFunction{};
|
||||
|
||||
switch (AtomicOp) {
|
||||
@@ -2089,12 +2073,21 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
case ExclusiveAtomicPairType::TYPE_AND:
|
||||
DesiredFunction = ANDDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_BIC:
|
||||
DesiredFunction = BICDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_OR:
|
||||
DesiredFunction = ORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_ORN:
|
||||
DesiredFunction = ORNDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EOR:
|
||||
DesiredFunction = EORDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_EON:
|
||||
DesiredFunction = EONDesired;
|
||||
break;
|
||||
case ExclusiveAtomicPairType::TYPE_NEG:
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
@@ -2140,7 +2133,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
@@ -2164,7 +2157,7 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
@@ -2186,6 +2179,60 @@ bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t LDUR = 0b0011'1000'0100'0000'0000'0000'0000'0000;
|
||||
LDUR |= Size << 30;
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint32_t STUR = 0b0011'1000'0000'0000'0000'0000'0000'0000;
|
||||
STUR |= Size << 30;
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXP_MASK) == FEXCore::ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
//Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleCASPAL_ARMv8(ucontext, info, Instr);
|
||||
|
||||
+22
-9
@@ -12,6 +12,10 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
|
||||
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
|
||||
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
|
||||
|
||||
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
|
||||
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
|
||||
|
||||
@@ -27,13 +31,19 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
|
||||
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
|
||||
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t EON_INST = 0x4A'20'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
@@ -46,8 +56,11 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
TYPE_ADD,
|
||||
TYPE_SUB,
|
||||
TYPE_AND,
|
||||
TYPE_BIC,
|
||||
TYPE_OR,
|
||||
TYPE_ORN,
|
||||
TYPE_EOR,
|
||||
TYPE_EON,
|
||||
TYPE_NEG, // This is just a sub with zero. Need to know the differences
|
||||
};
|
||||
|
||||
@@ -83,8 +96,8 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -16,7 +18,9 @@ namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
@@ -42,12 +46,57 @@ void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant,
|
||||
}
|
||||
|
||||
int NumMoves = 1;
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
// We only want to use ADRP+ADD if we have more than 1 segment
|
||||
for (size_t i = 0; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
if (Part != 0) {
|
||||
++RequiredMoveSegments;
|
||||
}
|
||||
}
|
||||
|
||||
// ADRP+ADD is specifically optimized in hardware
|
||||
// Check if we can use this
|
||||
auto PC = GetCursorAddress<uint64_t>();
|
||||
|
||||
// PC aligned to page
|
||||
uint64_t AlignedPC = PC & ~0xFFFULL;
|
||||
|
||||
// Offset from aligned PC
|
||||
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
}
|
||||
else {
|
||||
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
|
||||
// 21-bit signed integer here
|
||||
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
|
||||
if (vixl::IsInt21(SmallOffset)) {
|
||||
adr(Reg, SmallOffset);
|
||||
}
|
||||
else {
|
||||
// Need to use ADRP + ADD
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
add(Reg, Reg, Constant & 0xFFF);
|
||||
NumMoves = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -276,8 +325,126 @@ void Arm64Emitter::ResetStack() {
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Arm64Emitter::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return Dispatcher->ExitFunctionLinkerAddress;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64Emitter::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint64_t>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64Emitter::NamedSymbolLiteralPair Arm64Emitter::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64Emitter::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64Emitter::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint64_t>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64Emitter::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint64_t>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64Emitter::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,5 +1,8 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
@@ -60,6 +63,9 @@ class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
@@ -79,8 +85,66 @@ protected:
|
||||
|
||||
void ResetStack();
|
||||
void Align16B();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint64_t GuestEntry{};
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
+1
-1
@@ -658,7 +658,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // Reserved
|
||||
(0 << 29) | // SHA instructions
|
||||
(1 << 29) | // SHA instructions
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
|
||||
@@ -1,165 +0,0 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
#include "FEXCore/HLE/Linux/ThreadManagement.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <memory>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
|
||||
namespace FEXCore {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CompileService *This = reinterpret_cast<FEXCore::CompileService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
CompileService::CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ParentThread {Thread} {
|
||||
|
||||
CompileThreadData = std::make_unique<FEXCore::Core::InternalThreadState>();
|
||||
CompileThreadData->IsCompileService = true;
|
||||
|
||||
// We need a compiler for this work thread
|
||||
CTX->InitializeCompiler(CompileThreadData.get(), true);
|
||||
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CompileService::Initialize() {
|
||||
// Share CompileService which = this
|
||||
CompileThreadData->CompileService = ParentThread->CompileService;
|
||||
}
|
||||
|
||||
void CompileService::Shutdown() {
|
||||
ShuttingDown = true;
|
||||
// Kick the working thread
|
||||
StartWork.NotifyAll();
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
|
||||
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
|
||||
if (CompileMutex.try_lock()) {
|
||||
// We can only clear these things if we pulled the compile mutex
|
||||
|
||||
// Grab the work queue and clear it
|
||||
// We don't need to grab the queue mutex since this thread will no longer receive any work events
|
||||
// Threads are bounded 1:1
|
||||
while (!WorkQueue.empty()) {
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Go through the garbage collection array and clear it
|
||||
// It's safe to clear things that aren't marked safe since we are clearing cache
|
||||
GCArray.clear();
|
||||
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LocalIRCache.empty(), "Compile service must never have LocalIRCache");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
|
||||
// Clear the inverse cache of what is calling us from the Context ClearCache routine
|
||||
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
|
||||
SelectedThread->LookupCache->ClearCache();
|
||||
SelectedThread->CPUBackend->ClearCache();
|
||||
}
|
||||
|
||||
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
|
||||
WorkItem* ResultItem = nullptr;
|
||||
|
||||
{
|
||||
// Tell the worker thread to compile code for us
|
||||
auto Item = std::make_unique<WorkItem>();
|
||||
Item->RIP = RIP;
|
||||
|
||||
// Fill the threads work queue
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
ResultItem = WorkQueue.emplace(std::move(Item)).get();
|
||||
}
|
||||
|
||||
// Notify the thread that it has more work
|
||||
StartWork.NotifyAll();
|
||||
|
||||
return ResultItem;
|
||||
}
|
||||
|
||||
void CompileService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16]{};
|
||||
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
|
||||
while (true) {
|
||||
// Wait for work
|
||||
StartWork.Wait();
|
||||
if (ShuttingDown.load()) {
|
||||
break;
|
||||
}
|
||||
|
||||
std::scoped_lock lk(CompileMutex);
|
||||
size_t WorkItems{};
|
||||
|
||||
do {
|
||||
// Grab a work item
|
||||
std::unique_ptr<WorkItem> Item{};
|
||||
{
|
||||
std::scoped_lock lk(QueueMutex);
|
||||
WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
Item = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
}
|
||||
|
||||
// If we had a work item then work on it
|
||||
if (Item) {
|
||||
// Make sure it's not in lookup cache by accident
|
||||
LOGMAN_THROW_A_FMT(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->CurrentFrame->State.rip = Item->RIP;
|
||||
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
if (!CodePtr) {
|
||||
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
|
||||
ERROR_AND_DIE_FMT("Couldn't compile code for thread at RIP: 0x{:x}", Item->RIP);
|
||||
}
|
||||
|
||||
Item->CodePtr = CodePtr;
|
||||
Item->IRList = IRList;
|
||||
Item->DebugData = DebugData;
|
||||
Item->RAData = RAData;
|
||||
Item->StartAddr = StartAddr;
|
||||
Item->Length = Length;
|
||||
|
||||
auto& GCItem = GCArray.emplace_back(std::move(Item));
|
||||
GCItem->ServiceWorkDone.NotifyAll();
|
||||
}
|
||||
} while (WorkItems != 0);
|
||||
|
||||
// Clean up any safe entries in our GC array if we have any.
|
||||
std::erase_if(GCArray, [](const auto& Entry) {
|
||||
return Entry->SafeToClear.load(std::memory_order_relaxed);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,68 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <stdint.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
};
|
||||
class CompileService final {
|
||||
public:
|
||||
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
void Initialize();
|
||||
void Shutdown();
|
||||
|
||||
struct WorkItem {
|
||||
// Incoming
|
||||
uint64_t RIP{};
|
||||
|
||||
// Outgoing
|
||||
void *CodePtr{};
|
||||
FEXCore::IR::IRListView *IRList{};
|
||||
FEXCore::IR::RegisterAllocationData *RAData{};
|
||||
FEXCore::Core::DebugData *DebugData{};
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
|
||||
// Communication
|
||||
Event ServiceWorkDone{};
|
||||
std::atomic_bool SafeToClear{};
|
||||
};
|
||||
|
||||
WorkItem *CompileCode(uint64_t RIP);
|
||||
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address) const {
|
||||
return CompileThreadData->CPUBackend->IsAddressInJITCode(Address, false, false);
|
||||
}
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ParentThread;
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::unique_ptr<FEXCore::Core::InternalThreadState> CompileThreadData;
|
||||
|
||||
std::mutex QueueMutex{};
|
||||
std::mutex CompileMutex{};
|
||||
std::queue<std::unique_ptr<WorkItem>> WorkQueue{};
|
||||
std::vector<std::unique_ptr<WorkItem>> GCArray{};
|
||||
Event StartWork{};
|
||||
std::atomic_bool ShuttingDown{false};
|
||||
};
|
||||
}
|
||||
+138
-83
@@ -9,11 +9,11 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterCore.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
@@ -42,6 +42,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -147,10 +148,17 @@ namespace FEXCore::Context {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = std::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
}
|
||||
|
||||
Context::~Context() {
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
CodeObjectCacheService->Shutdown();
|
||||
}
|
||||
|
||||
for (auto &Thread : Threads) {
|
||||
if (Thread->ExecutionThread->joinable()) {
|
||||
Thread->ExecutionThread->join(nullptr);
|
||||
@@ -158,10 +166,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
for (auto &Thread : Threads) {
|
||||
|
||||
if (Thread->CompileService) {
|
||||
Thread->CompileService->Shutdown();
|
||||
}
|
||||
delete Thread;
|
||||
}
|
||||
Threads.clear();
|
||||
@@ -348,6 +352,9 @@ namespace FEXCore::Context {
|
||||
// Walk the threads and tell them to clear their caches
|
||||
// Useful when our block size is set to a large number and we need to step a single instruction
|
||||
for (auto &Thread : Threads) {
|
||||
// Wait for thread to be fully constructed
|
||||
// XXX: Look into thread partial construction issues
|
||||
while(Thread->RunningEvents.WaitingToStart.load()) ;
|
||||
ClearCodeCache(Thread, true);
|
||||
}
|
||||
}
|
||||
@@ -452,6 +459,7 @@ namespace FEXCore::Context {
|
||||
: nullptr),
|
||||
decltype(Entry.DebugData)(new Core::DebugData())
|
||||
};
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->LocalIRCache.insert({Addr, std::move(Entry)});
|
||||
};
|
||||
|
||||
@@ -494,7 +502,7 @@ namespace FEXCore::Context {
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
|
||||
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread) {
|
||||
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* State) {
|
||||
State->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
State->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
State->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
|
||||
@@ -512,7 +520,7 @@ namespace FEXCore::Context {
|
||||
bool DoSRA = false;
|
||||
#endif
|
||||
|
||||
State->PassManager->AddDefaultPasses(Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
|
||||
State->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
|
||||
State->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
State->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
@@ -521,16 +529,16 @@ namespace FEXCore::Context {
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread);
|
||||
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State);
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
State->PassManager->InsertRegisterAllocationPass(DoSRA);
|
||||
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
State->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, State, CompileThread);
|
||||
State->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, State);
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
State->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, State, CompileThread);
|
||||
State->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, State);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
#endif
|
||||
@@ -562,7 +570,7 @@ namespace FEXCore::Context {
|
||||
// Set up the thread manager state
|
||||
Thread->ThreadManager.parent_tid = ParentTID;
|
||||
|
||||
InitializeCompiler(Thread, false);
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
return Thread;
|
||||
@@ -624,24 +632,25 @@ namespace FEXCore::Context {
|
||||
|
||||
// Clean up dead stacks
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
|
||||
if (LiveThread->CompileService) {
|
||||
// If this live thread had a compile service then it no longer exists
|
||||
// Erase the shared_ptr
|
||||
LiveThread->CompileService.reset();
|
||||
}
|
||||
}
|
||||
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr, Start, Length);
|
||||
// Only call MarkGuestExecutableRange if new pages are marked as containing code
|
||||
if (Thread->LookupCache->AddBlockMapping(Address, Ptr, Start, Length)) {
|
||||
Thread->CTX->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
}
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache) {
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LookupCache->ClearCache();
|
||||
Thread->CPUBackend->ClearCache();
|
||||
if (Thread->CompileService) {
|
||||
Thread->CompileService->ClearCache(Thread);
|
||||
}
|
||||
|
||||
if (AlsoClearIRCache) {
|
||||
Thread->LocalIRCache.clear();
|
||||
@@ -677,15 +686,16 @@ namespace FEXCore::Context {
|
||||
}
|
||||
};
|
||||
|
||||
static void ValidateIR(FEXCore::Core::InternalThreadState *Thread) {
|
||||
static void ValidateIR(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction();
|
||||
static auto compaction = IR::CreateIRCompaction(ctx->OpDispatcherAllocator);
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
auto reparsed = IR::Parse(&out);
|
||||
FEXCore::Utils::PooledAllocatorMalloc Allocator;
|
||||
auto reparsed = IR::Parse(Allocator, &out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
@@ -713,6 +723,8 @@ namespace FEXCore::Context {
|
||||
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks);
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
@@ -749,7 +761,7 @@ namespace FEXCore::Context {
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_RemoveCodeEntry();
|
||||
Thread->OpDispatcher->_RemoveThreadCodeEntry();
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
@@ -813,7 +825,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
ValidateIR(Thread);
|
||||
ValidateIR(this, Thread);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -837,7 +849,8 @@ namespace FEXCore::Context {
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
auto IRList = Thread->OpDispatcher->CreateIRCopy();
|
||||
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
.IRList = IRList,
|
||||
@@ -857,6 +870,7 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
// Do we already have this in the IR cache?
|
||||
auto LocalEntry = Thread->LocalIRCache.find(GuestRIP);
|
||||
|
||||
@@ -872,6 +886,25 @@ namespace FEXCore::Context {
|
||||
GeneratedIR = false;
|
||||
}
|
||||
|
||||
// JIT Code object cache lookup
|
||||
if (CodeObjectCacheService) {
|
||||
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
|
||||
if (CodeCacheEntry) {
|
||||
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = CompiledCode,
|
||||
.IRData = nullptr, // No IR data generated
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.RAData = nullptr, // No RA data generated
|
||||
.GeneratedIR = false, // nullptr here ensures IR cache mechanisms won't run
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
@@ -933,6 +966,9 @@ namespace FEXCore::Context {
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Needs to be held for SMC interactions around concurrent compile and invalidation hazards
|
||||
auto InvalidationLk = Thread->CTX->SyscallHandler->CompileCodeLock(GuestRIP);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
@@ -944,63 +980,66 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
|
||||
bool DecrementRefCount = false;
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
if (Thread->CompileBlockReentrantRefCount != 0) {
|
||||
if (!Thread->CompileService) {
|
||||
Thread->CompileService = std::make_shared<FEXCore::CompileService>(this, Thread);
|
||||
Thread->CompileService->Initialize();
|
||||
}
|
||||
|
||||
auto* WorkItem = Thread->CompileService->CompileCode(GuestRIP);
|
||||
WorkItem->ServiceWorkDone.Wait();
|
||||
// Return here with the data in place
|
||||
CodePtr = WorkItem->CodePtr;
|
||||
IRList = WorkItem->IRList;
|
||||
DebugData = WorkItem->DebugData;
|
||||
RAData = WorkItem->RAData;
|
||||
StartAddr = WorkItem->StartAddr;
|
||||
Length = WorkItem->Length;
|
||||
WorkItem->SafeToClear = true;
|
||||
|
||||
// The compile service will always generate IR + DebugData + RAData
|
||||
// Remove the entries here to make sure we don't fail to insert later on
|
||||
RemoveCodeEntry(Thread, GuestRIP);
|
||||
GeneratedIR = true;
|
||||
} else {
|
||||
++Thread->CompileBlockReentrantRefCount;
|
||||
DecrementRefCount = true;
|
||||
auto [Code, IR, Data, RA, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
RAData = RA;
|
||||
GeneratedIR = Generated;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
}
|
||||
auto [Code, IR, Data, RA, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
RAData = RA;
|
||||
GeneratedIR = Generated;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
if (CodePtr == nullptr) {
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
return 0;
|
||||
}
|
||||
|
||||
// The core managed to compile the code.
|
||||
if (Config.BlockJITNaming()) {
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = this->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(CodePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
|
||||
} else {
|
||||
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(CodePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
|
||||
} else {
|
||||
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Tell the object cache service to serialize the code if enabled
|
||||
if (CodeObjectCacheService &&
|
||||
Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE &&
|
||||
DebugData) {
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(std::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(
|
||||
CodeSerialize::AsyncJobHandler::SerializationJobData {
|
||||
.GuestRIP = GuestRIP,
|
||||
.GuestCodeLength = Length,
|
||||
.GuestCodeHash = 0,
|
||||
.HostCodeBegin = CodePtr,
|
||||
.HostCodeLength = DebugData->HostCodeSize,
|
||||
.HostCodeHash = 0,
|
||||
.ThreadJobRefCount = &Thread->ObjectCacheRefCounter,
|
||||
.Relocations = std::move(*DebugData->Relocations),
|
||||
}
|
||||
));
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(
|
||||
Thread,
|
||||
CodePtr,
|
||||
@@ -1010,15 +1049,11 @@ namespace FEXCore::Context {
|
||||
RAData,
|
||||
IRList,
|
||||
DebugData,
|
||||
GeneratedIR,
|
||||
DecrementRefCount)) {
|
||||
GeneratedIR)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
// Insert to lookup cache
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr, StartAddr, Length);
|
||||
|
||||
@@ -1053,8 +1088,15 @@ namespace FEXCore::Context {
|
||||
Thread->RunningEvents.Running = false;
|
||||
}
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
// XXX: This doesn't make sense when the parent thread doesn't outlive its children
|
||||
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
|
||||
|
||||
if (Thread->ThreadManager.parent_tid == 0) {
|
||||
CoreShuttingDown.store(true);
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
@@ -1075,21 +1117,32 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
if (Thread->CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_MMAN) {
|
||||
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
|
||||
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length) >> 12);
|
||||
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
|
||||
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (auto Address: it->second)
|
||||
Context::RemoveCodeEntry(Thread, Address);
|
||||
it->second.clear();
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (auto Address: it->second) {
|
||||
Context::RemoveThreadCodeEntry(Thread, Address);
|
||||
}
|
||||
it->second.clear();
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::mutex> lk(CTX->ThreadCreationMutex);
|
||||
for (auto &Thread : CTX->Threads) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Context::RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void Context::RemoveThreadCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LocalIRCache.erase(GuestRIP);
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
@@ -1100,7 +1153,7 @@ namespace FEXCore::Context {
|
||||
Thread->CurrentFrame->State.rip = RIP;
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
RemoveCodeEntry(Thread, RIP);
|
||||
RemoveThreadCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CompileBlock(Thread->CurrentFrame, RIP);
|
||||
@@ -1117,6 +1170,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
bool Context::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
std::lock_guard<std::recursive_mutex> lk(ParentThread->LookupCache->WriteLock);
|
||||
auto it = ParentThread->LocalIRCache.find(RIP);
|
||||
if (it == ParentThread->LocalIRCache.end()) {
|
||||
return false;
|
||||
@@ -1142,15 +1196,16 @@ namespace FEXCore::Context {
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Context::AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
IRCaptureCache.AddNamedRegion(Base, Size, Offset, filename);
|
||||
IR::AOTIRCacheEntry *Context::LoadAOTIRCacheEntry(const std::string &filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
return rv;
|
||||
}
|
||||
|
||||
void Context::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
IRCaptureCache.RemoveNamedRegion(Base, Size);
|
||||
void Context::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
|
||||
@@ -39,7 +39,7 @@ static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
: FEXCore::CPU::Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
SetAllowAssembler(true);
|
||||
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -759,7 +758,7 @@ void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher, bool IncludeCompileService) const {
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) const {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
@@ -770,9 +769,6 @@ bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher, bo
|
||||
return true;
|
||||
}
|
||||
|
||||
if (IncludeCompileService && ThreadState->CompileService && ThreadState->CompileService->IsAddressInJITCode(Address)) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -77,7 +77,7 @@ public:
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const;
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const;
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
@@ -193,10 +193,37 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
mov(rsi, STATE);
|
||||
@@ -204,12 +231,57 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
call(rax);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, config.ExitFunctionLinkThis);
|
||||
mov(rsi, STATE);
|
||||
@@ -217,7 +289,28 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
mov(rax, config.ExitFunctionLink);
|
||||
call(rax);
|
||||
jmp(rax);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else {
|
||||
jmp(rax);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
|
||||
+4
-8
@@ -177,17 +177,12 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
|
||||
Decoder::Decoder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN } {
|
||||
// Using mmap is a start-up time optimization
|
||||
// Take advantage of page faulting to reduce startup time for minimal runtime cost
|
||||
DecodedBuffer =
|
||||
reinterpret_cast<FEXCore::X86Tables::DecodedInst *>(
|
||||
FEXCore::Allocator::mmap(0, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize,
|
||||
PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN }
|
||||
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
}
|
||||
|
||||
Decoder::~Decoder() {
|
||||
FEXCore::Allocator::munmap(DecodedBuffer, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize);
|
||||
PoolObject.UnclaimBuffer();
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
@@ -1148,6 +1143,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// XXX: Load symbol data
|
||||
SymbolAvailable = false;
|
||||
|
||||
@@ -38,6 +38,11 @@ public:
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
private:
|
||||
// To pass any information from instruction prefixes
|
||||
// down into the actual instruction handling machinery.
|
||||
@@ -63,6 +68,7 @@ private:
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
uint8_t const *InstStream;
|
||||
|
||||
+4
-5
@@ -59,11 +59,8 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
@@ -83,6 +80,8 @@ HostFeatures::HostFeatures() {
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
|
||||
@@ -19,6 +19,7 @@ class HostFeatures final {
|
||||
bool SupportsCLZERO{};
|
||||
bool SupportsAtomics{};
|
||||
bool SupportsRCPC{};
|
||||
bool SupportsTSOImm9{};
|
||||
bool SupportsRAND{};
|
||||
|
||||
// Float exception behaviour
|
||||
|
||||
@@ -147,8 +147,8 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
Data->State->CTX->RemoveCodeEntry(Data->State, Data->CurrentEntry);
|
||||
DEF_OP(RemoveThreadCodeEntry) {
|
||||
Data->State->CTX->RemoveThreadCodeEntry(Data->State, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
|
||||
@@ -356,6 +356,81 @@ DEF_OP(F80BCDSTORE) {
|
||||
memcpy(GDP, BCD, 10);
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = sin(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = cos(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = tan(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64F2XM1>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = exp2(Src) - 1.0;
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64ATAN>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = atan2(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = fmod(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = remainder(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F64FYL2X>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = Src2 * log2(Src1);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double trunc = (double)(int64_t)(Src2); //truncate
|
||||
double Tmp = Src1 * exp2(trunc);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -222,6 +222,73 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(double src) {
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(double src) {
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(double src) {
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(double src) {
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(double src1, double src2) {
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(double src1, double src2) {
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(double src1, double src2) {
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(double src1, double src2) {
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(double src1, double src2) {
|
||||
double trunc = (double)(int64_t)(src2); //truncate
|
||||
return src1 * exp2(trunc);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
|
||||
@@ -21,8 +21,7 @@ using DestMapType = std::vector<uint32_t>;
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
|
||||
|
||||
@@ -34,14 +34,11 @@ static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
InterpreterOps::InterpretIR(Thread, Thread->CurrentFrame->State.rip, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
}
|
||||
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, State {Thread} {
|
||||
|
||||
if (!CompileThread &&
|
||||
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
CreateAsmDispatch(ctx, Thread);
|
||||
}
|
||||
CreateAsmDispatch(ctx, Thread);
|
||||
}
|
||||
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
@@ -70,8 +67,8 @@ void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR:
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<InterpreterCore>(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
@@ -14,8 +14,7 @@ namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
|
||||
@@ -47,6 +47,16 @@ FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackH
|
||||
return {FABI_F64_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I16_F80, (void*)fn, HandlerIndex};
|
||||
@@ -122,6 +132,18 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
|
||||
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
|
||||
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
|
||||
|
||||
// Double Precision
|
||||
Info[Core::OPINDEX_F64SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle, Core::OPINDEX_F64SIN).fn);
|
||||
Info[Core::OPINDEX_F64COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle, Core::OPINDEX_F64COS).fn);
|
||||
Info[Core::OPINDEX_F64TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle, Core::OPINDEX_F64TAN).fn);
|
||||
Info[Core::OPINDEX_F64ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle, Core::OPINDEX_F64ATAN).fn);
|
||||
Info[Core::OPINDEX_F64F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle, Core::OPINDEX_F64F2XM1).fn);
|
||||
Info[Core::OPINDEX_F64FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle, Core::OPINDEX_F64FYL2X).fn);
|
||||
Info[Core::OPINDEX_F64FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle, Core::OPINDEX_F64FPREM).fn);
|
||||
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
|
||||
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
@@ -238,6 +260,12 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
@@ -261,6 +289,19 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
// Double Precision Unary
|
||||
COMMON_F64_OP(F2XM1)
|
||||
COMMON_F64_OP(TAN)
|
||||
COMMON_F64_OP(SIN)
|
||||
COMMON_F64_OP(COS)
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_F64_OP(FYL2X)
|
||||
COMMON_F64_OP(ATAN)
|
||||
COMMON_F64_OP(FPREM1)
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -123,7 +123,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(REMOVETHREADCODEENTRY, RemoveThreadCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
|
||||
// Conversion ops
|
||||
@@ -176,6 +176,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
@@ -311,6 +312,17 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(F80BCDLOAD, F80BCDLOAD);
|
||||
REGISTER_OP(F80BCDSTORE, F80BCDSTORE);
|
||||
|
||||
// F64 ops
|
||||
REGISTER_OP(F64SIN, F64SIN);
|
||||
REGISTER_OP(F64COS, F64COS);
|
||||
REGISTER_OP(F64TAN, F64TAN);
|
||||
REGISTER_OP(F64F2XM1, F64F2XM1);
|
||||
REGISTER_OP(F64ATAN, F64ATAN);
|
||||
REGISTER_OP(F64FPREM, F64FPREM);
|
||||
REGISTER_OP(F64FPREM1, F64FPREM1);
|
||||
REGISTER_OP(F64FYL2X, F64FYL2X);
|
||||
REGISTER_OP(F64SCALE, F64SCALE);
|
||||
|
||||
return Handlers;
|
||||
}();
|
||||
|
||||
|
||||
@@ -28,6 +28,8 @@ namespace FEXCore::CPU {
|
||||
FABI_F80_I32,
|
||||
FABI_F32_F80,
|
||||
FABI_F64_F80,
|
||||
FABI_F64_F64,
|
||||
FABI_F64_F64_F64,
|
||||
FABI_I16_F80,
|
||||
FABI_I32_F80,
|
||||
FABI_I64_F80,
|
||||
@@ -151,7 +153,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(RemoveThreadCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -197,6 +199,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -328,6 +331,17 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(F80CMP);
|
||||
DEF_OP(F80BCDLOAD);
|
||||
DEF_OP(F80BCDSTORE);
|
||||
|
||||
//< F64 ops
|
||||
DEF_OP(F64SIN);
|
||||
DEF_OP(F64COS);
|
||||
DEF_OP(F64TAN);
|
||||
DEF_OP(F64F2XM1);
|
||||
DEF_OP(F64ATAN);
|
||||
DEF_OP(F64FPREM);
|
||||
DEF_OP(F64FPREM1);
|
||||
DEF_OP(F64FYL2X);
|
||||
DEF_OP(F64SCALE);
|
||||
#undef DEF_OP
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
|
||||
@@ -157,6 +157,11 @@ DEF_OP(RDRAND) {
|
||||
// Second result is if we managed to read a valid random number or not
|
||||
DstPtr[1] = Result == 8 ? 1 : 0;
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
// Nop implementation
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -509,10 +509,10 @@ DEF_OP(AtomicSwap) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
case 1: swpalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swpalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -73,7 +73,7 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress};
|
||||
Literal l_BranchHost{Dispatcher->ExitFunctionLinkerAddress};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
|
||||
ldr(x0, &l_BranchHost);
|
||||
@@ -442,7 +442,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(RemoveThreadCodeEntry) {
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
@@ -452,7 +452,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Entry);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.RemoveCodeEntryFromJIT)));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.RemoveThreadCodeEntryFromJIT)));
|
||||
SpillStaticRegs();
|
||||
blr(x2);
|
||||
FillStaticRegs();
|
||||
@@ -500,7 +500,7 @@ void Arm64JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(REMOVETHREADCODEENTRY, RemoveThreadCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
+47
-23
@@ -71,11 +71,6 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
}
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
@@ -203,6 +198,43 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
@@ -357,7 +389,7 @@ void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
@@ -411,15 +443,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
}
|
||||
|
||||
if (!CompileThread) {
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalReturnInstruction = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
ThreadSharedData.OverflowExceptionInstructionAddress = Dispatcher->OverflowExceptionInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
}
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
@@ -430,7 +453,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
Pointers.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Pointers.RemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit);
|
||||
Pointers.RemoveThreadCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveThreadCodeEntryFromJit);
|
||||
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
@@ -499,7 +522,7 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
auto Buffer = GetBuffer();
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (Dispatcher->SignalHandlerRefCounter == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
@@ -716,9 +739,9 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
auto GuestEntry = GetCursorAddress<uint64_t>();
|
||||
GuestEntry = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
@@ -807,6 +830,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
this->IR = nullptr;
|
||||
@@ -823,11 +847,11 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
|
||||
return core->Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
|
||||
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
@@ -863,8 +887,8 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
+7
-17
@@ -43,8 +43,7 @@ public:
|
||||
};
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -63,13 +62,14 @@ public:
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
|
||||
}
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
|
||||
@@ -175,17 +175,6 @@ private:
|
||||
|
||||
static uint64_t ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalReturnInstruction{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
@@ -291,7 +280,7 @@ private:
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(RemoveThreadCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -341,6 +330,7 @@ private:
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
|
||||
@@ -617,13 +617,43 @@ DEF_OP(LoadMem) {
|
||||
DEF_OP(LoadMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("LoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "LoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldapurb(Dst, MemSrc);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
@@ -744,13 +774,41 @@ DEF_OP(StoreMem) {
|
||||
|
||||
DEF_OP(StoreMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("StoreMemTSO: No offset allowed");
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "StoreMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlurh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlur(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlur(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
|
||||
@@ -219,6 +219,10 @@ DEF_OP(RDRAND) {
|
||||
cset(Dst.second, Condition::ne);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
hint(SystemHint::YIELD);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -237,6 +241,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
+2
-4
@@ -14,13 +14,11 @@ namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -91,7 +91,7 @@ DEF_OP(ExitFunction) {
|
||||
jmp(qword[rax]);
|
||||
|
||||
L(l_BranchHost);
|
||||
dq(ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress);
|
||||
dq(Dispatcher->ExitFunctionLinkerAddress);
|
||||
L(l_BranchGuest);
|
||||
dq(NewRIP);
|
||||
} else {
|
||||
@@ -259,7 +259,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
DEF_OP(RemoveThreadCodeEntry) {
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -272,7 +272,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
mov(rax, Entry); // imm64 move
|
||||
mov(rsi, rax);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.RemoveCodeEntryFromJIT)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.RemoveThreadCodeEntryFromJIT)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -330,7 +330,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(REMOVETHREADCODEENTRY, RemoveThreadCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
+45
-29
@@ -75,11 +75,6 @@ void FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
void X86JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
}
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
@@ -196,6 +191,33 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
movsd(xmm1, GetSrc(IROp->Args[1].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
@@ -298,7 +320,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread)
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer)
|
||||
: CodeGenerator(Buffer.Size, Buffer.Ptr, nullptr)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread}
|
||||
@@ -334,23 +356,14 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
if (!CompileThread) {
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
ThreadSharedData.OverflowExceptionInstructionAddress = Dispatcher->OverflowExceptionInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
}
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
@@ -358,7 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
// Process specific
|
||||
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Pointers.RemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit);
|
||||
Pointers.RemoveThreadCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveThreadCodeEntryFromJit);
|
||||
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
@@ -416,7 +429,7 @@ void X86JITCore::EmitDetectionString() {
|
||||
}
|
||||
|
||||
void X86JITCore::ClearCache() {
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (Dispatcher->SignalHandlerRefCounter == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
@@ -636,6 +649,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
}
|
||||
|
||||
void *GuestEntry = getCurr<void*>();
|
||||
CursorEntry = getSize();
|
||||
this->IR = IR;
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
@@ -650,7 +664,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(RunBlock);
|
||||
// Else we need to pause now
|
||||
mov(rax, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
|
||||
mov(rax, Dispatcher->ThreadPauseHandlerAddress);
|
||||
jmp(rax);
|
||||
ud2();
|
||||
|
||||
@@ -788,7 +802,9 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(GuestExit) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
@@ -800,10 +816,10 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
|
||||
return core->Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
|
||||
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
@@ -813,8 +829,8 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, X86JITCore::INITIAL_CODE_SIZE));
|
||||
}
|
||||
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
+69
-17
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
@@ -58,8 +59,7 @@ class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
CodeBuffer Buffer,
|
||||
bool CompileThread);
|
||||
CodeBuffer Buffer);
|
||||
~X86JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -77,15 +77,77 @@ public:
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
|
||||
}
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
void LoadConstantWithPadding(Xbyak::Reg Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Label Offset;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(Xbyak::Reg Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(Xbyak::Reg Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint64_t CursorEntry{};
|
||||
/** @} */
|
||||
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
@@ -169,17 +231,6 @@ private:
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
@@ -289,7 +340,7 @@ private:
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(RemoveThreadCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -334,6 +385,7 @@ private:
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
|
||||
@@ -166,6 +166,10 @@ DEF_OP(RDRAND) {
|
||||
setc(Dst.second.cvt8());
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
pause();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -184,6 +188,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|x86-64
|
||||
desc: relocation logic of the x86-64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
uint64_t X86JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return Dispatcher->ExitFunctionLinkerAddress;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
|
||||
}
|
||||
|
||||
void X86JITCore::LoadConstantWithPadding(Xbyak::Reg Reg, uint64_t Constant) {
|
||||
// The maximum size a move constant can be in bytes
|
||||
// Need to NOP pad to this size to ensure backpatching is always the same size
|
||||
// Calculated as:
|
||||
// [Rex]
|
||||
// [Mov op]
|
||||
// [8 byte constant]
|
||||
//
|
||||
// All other move types are smaller than this. xbyak will use a NOP slide which is quite quick
|
||||
constexpr static size_t MAX_MOVE_SIZE = 10;
|
||||
auto StartingOffset = getSize();
|
||||
mov(Reg, Constant);
|
||||
auto MoveSize = getSize() - StartingOffset;
|
||||
auto NOPPadSize = MAX_MOVE_SIZE - MoveSize;
|
||||
nop(NOPPadSize);
|
||||
}
|
||||
|
||||
X86JITCore::NamedSymbolLiteralPair X86JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
NamedSymbolLiteralPair Lit {
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void X86JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = getSize();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CursorEntry;
|
||||
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Lit.MoveABI.NamedSymbolLiteral.Symbol);
|
||||
|
||||
L(Lit.Offset);
|
||||
dq(Pointer);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
|
||||
void X86JITCore::InsertGuestRIPMove(Xbyak::Reg Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = getSize();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CursorEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.getIdx();
|
||||
|
||||
if (CTX->Config.CacheObjectCodeCompilation()) {
|
||||
LoadConstantWithPadding(Reg, Constant);
|
||||
}
|
||||
else {
|
||||
mov(Reg, Constant);
|
||||
}
|
||||
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool X86JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Place the pointer
|
||||
dq(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(CTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstantWithPadding(Xbyak::Reg64(Reloc->NamedThunkMove.RegisterIndex), Pointer);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
setSize(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstantWithPadding(Xbyak::Reg64(Reloc->GuestRIPMove.RegisterIndex), Pointer);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -61,6 +61,7 @@ void LookupCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8, MADV_DONTNEED);
|
||||
madvise(reinterpret_cast<void*>(PageMemory), CODE_SIZE, MADV_DONTNEED);
|
||||
@@ -68,6 +69,8 @@ void LookupCache::ClearL2Cache() {
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1
|
||||
madvise(reinterpret_cast<void*>(L1Pointer), L1_SIZE, MADV_DONTNEED);
|
||||
// Clear L2
|
||||
|
||||
+65
-44
@@ -7,6 +7,7 @@
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -28,31 +29,63 @@ public:
|
||||
uintptr_t End() { return 0; }
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
auto HostCode = FindCodePointerForAddress(Address);
|
||||
if (HostCode) {
|
||||
return HostCode;
|
||||
} else {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
// Try L1, no lock needed
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
} else {
|
||||
return 0;
|
||||
// L2 and L3 need to be locked
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize -1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address)
|
||||
{
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto InsertPoint =
|
||||
#endif
|
||||
BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LOGMAN_THROW_A_FMT(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
rv |= CodePages[CurrentPage].size() == 0;
|
||||
CodePages[CurrentPage].push_back(Address);
|
||||
}
|
||||
|
||||
@@ -61,10 +94,14 @@ public:
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
void Erase(uint64_t Address) {
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks.lower_bound({Address, 0});
|
||||
auto upper = BlockLinks.upper_bound({Address, UINTPTR_MAX});
|
||||
@@ -78,7 +115,10 @@ public:
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = L1Entry.HostCode = 0;
|
||||
L1Entry.GuestCode = 0;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
// Do full map
|
||||
@@ -101,6 +141,8 @@ public:
|
||||
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks.insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
@@ -116,8 +158,19 @@ public:
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::LocalIRCache,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
@@ -167,38 +220,6 @@ private:
|
||||
return PageMemory + NewBase;
|
||||
}
|
||||
|
||||
uintptr_t FindCodePointerForAddress(uint64_t Address) {
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize -1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == FullAddress)
|
||||
{
|
||||
L1Entry.GuestCode = FullAddress;
|
||||
return L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
}
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
|
||||
+84
@@ -0,0 +1,84 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie{};
|
||||
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock{};
|
||||
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
bool MultiBlock : 1;
|
||||
|
||||
// TSO enabled
|
||||
bool TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
bool ABILocalFlags : 1;
|
||||
|
||||
// ABI no PF unsafe optimization
|
||||
bool ABINoPF : 1;
|
||||
|
||||
// Static register allocation enabled
|
||||
bool SRA : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
bool ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
bool Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
bool x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
SRA == other.SRA &&
|
||||
ParanoidTSO == other.ParanoidTSO &&
|
||||
Is64BitMode == other.Is64BitMode &&
|
||||
SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(CodeObjectSerializationConfig const &other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash{};
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
Hash <<= 1; Hash |= other.SRA;
|
||||
Hash <<= 1; Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1; Hash |= other.Is64BitMode;
|
||||
Hash <<= 2; Hash |= other.SMCChecks;
|
||||
Hash <<= 1; Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <sys/uio.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
auto BaseFilename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = std::make_unique<CodeRegionEntry>(
|
||||
Base,
|
||||
Size,
|
||||
Offset,
|
||||
filename,
|
||||
NamedRegionHandler->DefaultCodeHeader(Base, Offset)
|
||||
);
|
||||
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
std::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
}
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::Context *ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ABINoPF = ctx->Config.ABINoPF;
|
||||
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion *>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion *>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService *This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler { &NamedRegionHandler , this }
|
||||
, NamedRegionHandler { ctx } {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
WorkerThreadShuttingDown = true;
|
||||
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, std::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
CodeObjectFileSection const *CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16] = "ObjectCodeSeri\0";
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto &it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,459 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
#include "Interface/Core/ObjectCache/CodeObjectSerializationConfig.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <shared_mutex>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// XXX: Does this need to be signal safe?
|
||||
using CodeSerializationMutex = std::shared_mutex;
|
||||
struct CodeSerializationData {
|
||||
};
|
||||
|
||||
struct CodeObjectFileSection {
|
||||
bool Serialized;
|
||||
bool Invalid;
|
||||
const CodeSerializationData *Data;
|
||||
const char *HostCode;
|
||||
uint64_t NumRelocations;
|
||||
const char *Relocations;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is the file header that lives at the start of an object cache file
|
||||
*
|
||||
* This header is updated from multiple processes!
|
||||
* Care must be taken to use OS locks when updating the file backing including this header
|
||||
*/
|
||||
struct CodeObjectSerializationHeader {
|
||||
// The configuration that this file has
|
||||
CodeObjectSerializationConfig Config;
|
||||
// The original RIP that this object section was mapped at
|
||||
uint64_t OriginalBase{};
|
||||
// The original offset in to the file that this object section was loaded from
|
||||
uint64_t OriginalOffset{};
|
||||
// Total amount of code that should be in this file
|
||||
uint64_t TotalCodeSize{};
|
||||
// Used to reserve the TSL map
|
||||
uint64_t NumCodeEntries{};
|
||||
// The number of relocations that point to this section
|
||||
uint64_t NumRelocationsTo{};
|
||||
// Total relocations in this file
|
||||
uint64_t TotalRelocationsCount{};
|
||||
};
|
||||
|
||||
struct CodeRegionEntry {
|
||||
/**
|
||||
* @name Threaded initialization objects for the initial object creation
|
||||
* @{ */
|
||||
// Base address in memory where the code region is at
|
||||
uint64_t Base{};
|
||||
|
||||
// Size of this code entry
|
||||
uint64_t Size{};
|
||||
|
||||
// The offset inside the file that is mapped to Base
|
||||
uint64_t Offset{};
|
||||
|
||||
// Filename of the object
|
||||
std::string Filename{};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader{};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
std::string ObjectEntrySourceFilename{};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
bool StillSerializing {true};
|
||||
|
||||
// Long lived FD for serialization if we have multiple jobs to serialize
|
||||
// Bursts of code entries are common and this reduces file lock overhead
|
||||
//
|
||||
// Especially useful over network mounts where file locks are very slow
|
||||
int CurrentSerializedFD {-1};
|
||||
|
||||
/**
|
||||
* @name Objects required to sync objects between threads
|
||||
* @{ */
|
||||
// Refcount for the number of outstanding code entries waiting to be written for this object section
|
||||
CodeSerializationMutex ObjectJobRefCountMutex;
|
||||
|
||||
// Refcount for outstanding named object region entry loading itself
|
||||
// Will block JIT code cache look up when this has a unique_lock held
|
||||
CodeSerializationMutex NamedJobRefCountMutex;
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Object Entry data management
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @name This is the raw file data that we loaded from the code region entry file
|
||||
* @{ */
|
||||
char *CodeData{};
|
||||
size_t FileSize{};
|
||||
|
||||
std::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
tsl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
CodeRegionEntry() = default;
|
||||
|
||||
// Initializer specifically for threaded loading
|
||||
CodeRegionEntry(uint64_t Base,
|
||||
uint64_t Size,
|
||||
uint64_t Offset,
|
||||
std::string const &Filename,
|
||||
CodeObjectSerializationHeader const &DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
, Offset {Offset}
|
||||
, Filename {Filename}
|
||||
, EntryHeader {DefaultHeader} {
|
||||
}
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = std::map<uint64_t, std::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = std::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
|
||||
class AsyncJobHandler final {
|
||||
public:
|
||||
/**
|
||||
* @brief Structure containing all the data required to async serialize code objects
|
||||
*/
|
||||
struct SerializationJobData {
|
||||
uint64_t GuestRIP; ///< The RIP for the guest
|
||||
// XXX: Support multiblock
|
||||
uint64_t GuestCodeLength; ///< The Guest's code length
|
||||
uint64_t GuestCodeHash; ///< Hash of the guest code
|
||||
|
||||
void *HostCodeBegin; ///< Host JIT code starting memory address
|
||||
size_t HostCodeLength; ///< Host JIT code length
|
||||
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
|
||||
|
||||
// This is the thread specific ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
|
||||
// This way it will wait until the async job handler is complete with it.
|
||||
CodeSerializationMutex *ThreadJobRefCount;
|
||||
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
* @{ */
|
||||
// This is the code region's ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
|
||||
CodeSerializationMutex *ObjectJobRefCountMutexPtr;
|
||||
|
||||
// This is the code region iterator to reduce the number of map lookups
|
||||
// This will remain valid while jobs are outstanding for this region
|
||||
CodeRegionMapType::iterator CodeRegionIterator;
|
||||
/** @} */
|
||||
};
|
||||
|
||||
AsyncJobHandler(NamedRegionObjectHandler *NamedRegionHandler, CodeObjectSerializeService *CodeObjectCacheService)
|
||||
: NamedRegionHandler {NamedRegionHandler}
|
||||
, CodeObjectCacheService {CodeObjectCacheService} {}
|
||||
|
||||
protected:
|
||||
friend class CodeObjectSerializeService;
|
||||
friend class NamedRegionObjectHandler;
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Async named region handling
|
||||
* @{ */
|
||||
/**
|
||||
* @brief The async named region jobs to handle.
|
||||
*
|
||||
* Only two, Code serialization goes in to a different queue.
|
||||
*/
|
||||
enum class NamedRegionJobType {
|
||||
JOB_ADD_NAMED_REGION,
|
||||
JOB_REMOVE_NAMED_REGION,
|
||||
};
|
||||
|
||||
class NamedRegionWorkItem {
|
||||
public:
|
||||
NamedRegionJobType GetType() const { return Type; }
|
||||
|
||||
protected:
|
||||
friend class WorkItemAddNamedRegion;
|
||||
NamedRegionWorkItem(NamedRegionJobType type)
|
||||
: Type {type} {}
|
||||
|
||||
private:
|
||||
NamedRegionJobType Type;
|
||||
};
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const std::string &base, const std::string &filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry}
|
||||
{}
|
||||
const std::string BaseFilename;
|
||||
const std::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, std::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
, Entry {std::move(entry)} {}
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
std::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
private:
|
||||
NamedRegionObjectHandler *NamedRegionHandler;
|
||||
CodeObjectSerializeService *CodeObjectCacheService;
|
||||
};
|
||||
|
||||
class NamedRegionObjectHandler final {
|
||||
public:
|
||||
NamedRegionObjectHandler(FEXCore::Context::Context *ctx);
|
||||
|
||||
void HandleNamedRegionObjectJobs();
|
||||
|
||||
CodeObjectSerializationConfig const &GetDefaultSerializationConfig() const {
|
||||
return DefaultSerializationConfig;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
// Return a default code header based off the default serialization config
|
||||
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
|
||||
return CodeObjectSerializationHeader {
|
||||
.Config = DefaultSerializationConfig,
|
||||
.OriginalBase = Base,
|
||||
.OriginalOffset = Offset,
|
||||
.NumCodeEntries = 0,
|
||||
.NumRelocationsTo = 0,
|
||||
.TotalRelocationsCount = 0,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds an asynchronous add named region work item to the object queue
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const std::string &base, const std::string &filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(std::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
|
||||
base,
|
||||
filename,
|
||||
executable,
|
||||
entry
|
||||
));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, std::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(std::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
|
||||
Base,
|
||||
Size,
|
||||
std::move(Entry)
|
||||
));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
private:
|
||||
// Code version. If the code emission changes then this needs to increment
|
||||
constexpr static uint32_t CODE_VERSION = 0x0;
|
||||
|
||||
// Default cookie header for the file header
|
||||
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
|
||||
|
||||
// Code serialization config for our current process configuration
|
||||
CodeObjectSerializationConfig DefaultSerializationConfig;
|
||||
|
||||
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
|
||||
std::atomic<uint64_t> NamedWorkQueueJobs{};
|
||||
|
||||
// Mutex for ading new jobs to the work queue
|
||||
std::mutex NamedWorkQueueMutex{};
|
||||
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
std::queue<std::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const std::string &base_filename, const std::string &filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, std::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Context specific code object serialization class
|
||||
*
|
||||
* Contains everything required for FEXCore to serialize code objects
|
||||
*/
|
||||
class CodeObjectSerializeService final {
|
||||
public:
|
||||
CodeObjectSerializeService(FEXCore::Context::Context *ctx);
|
||||
|
||||
/**
|
||||
* @brief Initialize the internal interface
|
||||
*
|
||||
* Is a public interface to allow the service to reinitialize after forking
|
||||
*/
|
||||
void Initialize();
|
||||
|
||||
/**
|
||||
* @brief Safely shut down the Code Object serialization service.
|
||||
*
|
||||
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
|
||||
*/
|
||||
void Shutdown();
|
||||
|
||||
/**
|
||||
* @name Async interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Loads a named region in to the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address that this named region is loaded
|
||||
* @param Size - The size of the region
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Unloads a named region from the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address of the named region
|
||||
* @param Size - The size of the region
|
||||
*/
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds a code object serialization job. As async as possible.
|
||||
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(std::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Synchronous interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Synchronously waits for this thread's job queue to become empty.
|
||||
*
|
||||
* This is necessary for when a thread is shutting down
|
||||
*
|
||||
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
|
||||
*/
|
||||
static void WaitForEmptyJobQueue(CodeSerializationMutex *ThreadJobRefCount) {
|
||||
// Once the shared mutex is empty this unique lock will be gained
|
||||
std::unique_lock lk {*ThreadJobRefCount};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Fetches object code from the Code Object Cache for JIT.
|
||||
*
|
||||
* @param GuestRIP - Which GuestRIP to search the cache for
|
||||
*
|
||||
* @return Data required for the JIT to relocate the Object code.
|
||||
*/
|
||||
CodeObjectFileSection const *FetchCodeObjectFromCache(uint64_t GuestRIP);
|
||||
/** @} */
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
/**
|
||||
* @brief Safely closes out code object regions from the map
|
||||
*
|
||||
* @param it - iterator to do a closure on
|
||||
*/
|
||||
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it);
|
||||
|
||||
CodeSerializationMutex &GetEntryMapMutex() { return EntryMapMutex; }
|
||||
CodeSerializationMutex &GetUnrelocatedEntryMapMutex() { return EntryMapMutex; }
|
||||
|
||||
CodeRegionMapType &GetEntryMap() { return AddressToEntryMap; }
|
||||
CodeRegionPtrMapType &GetUnrelocatedEntryMap() { return UnrelocatedAddressToEntryMap; }
|
||||
|
||||
/**
|
||||
* @brief Notify the async thread that it has work to do
|
||||
*/
|
||||
void NotifyWork() { WorkAvailable.NotifyOne(); }
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
Event WorkAvailable{};
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
// Mutex to hold when modifying the entry maps
|
||||
CodeSerializationMutex EntryMapMutex;
|
||||
CodeSerializationMutex UnrelocatedEntryMapMutex;
|
||||
|
||||
// Entry maps
|
||||
CodeRegionMapType AddressToEntryMap;
|
||||
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
#pragma once
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum class RelocationTypes : uint8_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
RelocationTypes Type;
|
||||
};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
NamedSymbol Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
};
|
||||
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
};
|
||||
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
|
||||
// The unrelocated RIP that is being moved
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
};
|
||||
}
|
||||
+316
-31
@@ -1443,6 +1443,11 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// But this would result in a zext on 64bit, which would ruin the no-op nature of the instruction
|
||||
// So x86-64 spec mandates this special case that even though it is a 32bit instruction and
|
||||
// is supposed to zext the result, it is a true no-op
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) {
|
||||
// If this instruction has a REP prefix then this is architectually defined to be a `PAUSE` instruction.
|
||||
// On older processors this ends up being a true `REP NOP` which is why they stuck this here.
|
||||
_Yield();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -3353,7 +3358,9 @@ void OpDispatchBuilder::ReadSegmentReg(OpcodeArgs) {
|
||||
|
||||
template<OpDispatchBuilder::Segment Seg>
|
||||
void OpDispatchBuilder::WriteSegmentReg(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
// Documentation claims that the 32-bit version of this instruction inserts in to the lower 32-bits of the segment
|
||||
// This is incorrect and it instead zero extends the 32-bit value to 64-bit
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
if constexpr (Seg == Segment::FS) {
|
||||
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs));
|
||||
@@ -4604,7 +4611,7 @@ OrderedNode *OpDispatchBuilder::AppendSegmentOffset(OrderedNode *Value, uint32_t
|
||||
return Value;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad) {
|
||||
OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad, MemoryAccessType AccessType) {
|
||||
LOGMAN_THROW_A_FMT(Operand.IsGPR() ||
|
||||
Operand.IsLiteral() ||
|
||||
Operand.IsGPRDirect() ||
|
||||
@@ -4615,7 +4622,6 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
|
||||
OrderedNode *Src {nullptr};
|
||||
bool LoadableType = false;
|
||||
bool StackAccess = false;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const uint32_t AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
@@ -4644,7 +4650,9 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
else if (Operand.IsGPRDirect()) {
|
||||
Src = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPR.GPR]));
|
||||
LoadableType = true;
|
||||
StackAccess = Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsGPRIndirect()) {
|
||||
auto GPR = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPRIndirect.GPR]));
|
||||
@@ -4653,7 +4661,9 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
Src = _Add(GPR, Constant);
|
||||
|
||||
LoadableType = true;
|
||||
StackAccess = Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsRIPRelative()) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
@@ -4675,7 +4685,9 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
auto Constant = _Constant(GPRSize * 8, Operand.Data.SIB.Scale);
|
||||
Tmp = _Mul(Tmp, Constant);
|
||||
}
|
||||
StackAccess |= Operand.Data.SIB.Index == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.SIB.Index == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
|
||||
if (Operand.Data.SIB.Base != FEXCore::X86State::REG_INVALID) {
|
||||
@@ -4687,7 +4699,10 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
else {
|
||||
Tmp = GPR;
|
||||
}
|
||||
StackAccess |= Operand.Data.SIB.Base == FEXCore::X86State::REG_RSP;
|
||||
|
||||
if (Operand.Data.SIB.Base == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
|
||||
if (Operand.Data.SIB.Offset) {
|
||||
@@ -4722,7 +4737,7 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
if ((LoadableType && LoadData) || ForceLoad) {
|
||||
Src = AppendSegmentOffset(Src, Flags);
|
||||
|
||||
if (StackAccess) {
|
||||
if (AccessType == MemoryAccessType::ACCESS_NONTSO || AccessType == MemoryAccessType::ACCESS_STREAM) {
|
||||
Src = _LoadMem(Class, OpSize, Src, Align == -1 ? OpSize : Align);
|
||||
}
|
||||
else {
|
||||
@@ -4737,12 +4752,12 @@ OrderedNode *OpDispatchBuilder::GetRelocatedPC(FEXCore::X86Tables::DecodedOp con
|
||||
return _EntrypointOffset(Op->PC + Op->InstSize + Offset - Entry, GPRSize);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad) {
|
||||
OrderedNode *OpDispatchBuilder::LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad, MemoryAccessType AccessType) {
|
||||
const uint8_t OpSize = GetSrcSize(Op);
|
||||
return LoadSource_WithOpSize(Class, Op, Operand, OpSize, Flags, Align, LoadData, ForceLoad);
|
||||
return LoadSource_WithOpSize(Class, Op, Operand, OpSize, Flags, Align, LoadData, ForceLoad, AccessType);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align) {
|
||||
void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType) {
|
||||
LOGMAN_THROW_A_FMT(Operand.IsGPR() ||
|
||||
Operand.IsLiteral() ||
|
||||
Operand.IsGPRDirect() ||
|
||||
@@ -4755,7 +4770,6 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
// 32bit ops ZEXT the result to 64bit
|
||||
OrderedNode *MemStoreDst {nullptr};
|
||||
bool MemStore = false;
|
||||
bool StackAccess = false;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const uint32_t AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
@@ -4789,7 +4803,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
else if (Operand.IsGPRDirect()) {
|
||||
MemStoreDst = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPR.GPR]));
|
||||
MemStore = true;
|
||||
StackAccess = Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsGPRIndirect()) {
|
||||
auto GPR = _LoadContext(AddrSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[Operand.Data.GPRIndirect.GPR]));
|
||||
@@ -4797,7 +4813,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
|
||||
MemStoreDst = _Add(GPR, Constant);
|
||||
MemStore = true;
|
||||
StackAccess = Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP;
|
||||
if (Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::ACCESS_DEFAULT) {
|
||||
AccessType = MemoryAccessType::ACCESS_NONTSO;
|
||||
}
|
||||
}
|
||||
else if (Operand.IsRIPRelative()) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
@@ -4867,7 +4885,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
auto DestAddr = _Add(MemStoreDst, _Constant(8));
|
||||
_StoreMem(GPRClass, 2, DestAddr, Upper, std::min<uint8_t>(Align, 8));
|
||||
} else {
|
||||
if (StackAccess) {
|
||||
if (AccessType == MemoryAccessType::ACCESS_NONTSO || AccessType == MemoryAccessType::ACCESS_STREAM) {
|
||||
_StoreMem(Class, OpSize, MemStoreDst, Src, Align == -1 ? OpSize : Align);
|
||||
}
|
||||
else {
|
||||
@@ -4877,19 +4895,24 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align) {
|
||||
StoreResult_WithOpSize(Class, Op, Operand, Src, GetDstSize(Op), Align);
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType) {
|
||||
StoreResult_WithOpSize(Class, Op, Operand, Src, GetDstSize(Op), Align, AccessType);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align) {
|
||||
StoreResult(Class, Op, Op->Dest, Src, Align);
|
||||
void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType) {
|
||||
StoreResult(Class, Op, Op->Dest, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx} {
|
||||
: IREmitter {ctx->OpDispatcherAllocator}
|
||||
, CTX {ctx} {
|
||||
ResetWorkingList();
|
||||
InstallHostSpecificOpcodeHandlers();
|
||||
}
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
: IREmitter {Allocator}
|
||||
, CTX {nullptr} {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ResetWorkingList() {
|
||||
IREmitter::ResetWorkingList();
|
||||
@@ -4910,6 +4933,11 @@ void OpDispatchBuilder::MOVGPROp(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, 1);
|
||||
StoreResult(GPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
bool RequiresMask = false;
|
||||
FEXCore::IR::IROps IROp;
|
||||
@@ -5242,9 +5270,19 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
return;
|
||||
}
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F38_SHA[] = {
|
||||
{OPD(PF_38_NONE, 0xC8), 1, &OpDispatchBuilder::SHA1NEXTEOp},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, &OpDispatchBuilder::SHA1MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, &OpDispatchBuilder::SHA1MSG2Op},
|
||||
{OPD(PF_38_NONE, 0xCB), 1, &OpDispatchBuilder::SHA256RNDS2Op},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F38_AES[] = {
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
|
||||
@@ -5304,6 +5342,8 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38_CRC);
|
||||
}
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38_SHA);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAES) {
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38_AES);
|
||||
InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3A_AES);
|
||||
@@ -5447,7 +5487,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xBD, 1, &OpDispatchBuilder::BSROp}, // BSF
|
||||
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
|
||||
{0xC0, 2, &OpDispatchBuilder::XADDOp},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPROp<0>},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::PExtrOp<2>},
|
||||
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
|
||||
@@ -5461,7 +5501,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x17, 1, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
|
||||
@@ -5523,7 +5563,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xE3, 1, &OpDispatchBuilder::PAVGOp<2>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::PSUBSOp<1, true>},
|
||||
{0xE9, 1, &OpDispatchBuilder::PSUBSOp<2, true>},
|
||||
{0xEA, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMIN, 2>},
|
||||
@@ -5751,7 +5791,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<8>},
|
||||
@@ -5825,7 +5865,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorOp},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQSUB, 1>},
|
||||
{0xE9, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQSUB, 2>},
|
||||
{0xEA, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMIN, 2>},
|
||||
@@ -5920,10 +5960,6 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::FenceOp<FEXCore::IR::Fence_LoadStore.Val>}, //MFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, //SFENCE
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
@@ -5939,6 +5975,15 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_66, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryExtensionOpTable_64[] = {
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1, &OpDispatchBuilder::ReadSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1, &OpDispatchBuilder::WriteSegmentReg<OpDispatchBuilder::Segment::GS>},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryModRMExtensionOpTable[] = {
|
||||
@@ -5953,6 +5998,236 @@ constexpr uint16_t PF_F2 = 3;
|
||||
// All OPDReg versions need it
|
||||
#define OPDReg(op, reg) ((1 << 15) | ((op - 0xD8) << 8) | (reg << 3))
|
||||
#define OPD(op, modrmop) (((op - 0xD8) << 8) | modrmop)
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> X87F64OpTable[] = {
|
||||
{OPDReg(0xD8, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xD8, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xD8, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD8, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xD8, 0xC0), 8, &OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xC8), 8, &OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xD0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
{OPD(0xD8, 0xD8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
{OPD(0xD8, 0xE0), 8, &OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xE8), 8, &OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xF0), 8, &OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
{OPD(0xD8, 0xF8), 8, &OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xD9, 0) | 0x00, 8, &OpDispatchBuilder::FLDF64<32>},
|
||||
|
||||
// 1 = Invalid
|
||||
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::FSTF64<32>},
|
||||
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::FSTF64<32>},
|
||||
|
||||
{OPDReg(0xD9, 4) | 0x00, 8, &OpDispatchBuilder::X87LDENVF64},
|
||||
|
||||
{OPDReg(0xD9, 5) | 0x00, 8, &OpDispatchBuilder::X87FLDCWF64},
|
||||
|
||||
{OPDReg(0xD9, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSTENV},
|
||||
|
||||
{OPDReg(0xD9, 7) | 0x00, 8, &OpDispatchBuilder::X87FSTCW},
|
||||
|
||||
{OPD(0xD9, 0xC0), 8, &OpDispatchBuilder::FLDF64<80>},
|
||||
{OPD(0xD9, 0xC8), 8, &OpDispatchBuilder::FXCH},
|
||||
{OPD(0xD9, 0xD0), 1, &OpDispatchBuilder::NOPOp}, // FNOP
|
||||
// D1 = Invalid
|
||||
// D8 = Invalid
|
||||
{OPD(0xD9, 0xE0), 1, &OpDispatchBuilder::FCHSF64},
|
||||
{OPD(0xD9, 0xE1), 1, &OpDispatchBuilder::FABSF64},
|
||||
// E2 = Invalid
|
||||
{OPD(0xD9, 0xE4), 1, &OpDispatchBuilder::FTSTF64},
|
||||
{OPD(0xD9, 0xE5), 1, &OpDispatchBuilder::X87FXAMF64},
|
||||
// E6 = Invalid
|
||||
{OPD(0xD9, 0xE8), 1, &OpDispatchBuilder::FLDF64_Const<0x3FF0000000000000>}, // 1.0
|
||||
{OPD(0xD9, 0xE9), 1, &OpDispatchBuilder::FLDF64_Const<0x400A934F0979A372>}, // log2l(10)
|
||||
{OPD(0xD9, 0xEA), 1, &OpDispatchBuilder::FLDF64_Const<0x3FF71547652B82FE>}, // log2l(e)
|
||||
{OPD(0xD9, 0xEB), 1, &OpDispatchBuilder::FLDF64_Const<0x400921FB54442D18>}, // pi
|
||||
{OPD(0xD9, 0xEC), 1, &OpDispatchBuilder::FLDF64_Const<0x3FD34413509F79FF>}, // log10l(2)
|
||||
{OPD(0xD9, 0xED), 1, &OpDispatchBuilder::FLDF64_Const<0x3FE62E42FEFA39EF>}, // log(2)
|
||||
{OPD(0xD9, 0xEE), 1, &OpDispatchBuilder::FLDF64_Const<0>}, // 0.0
|
||||
|
||||
// EF = Invalid
|
||||
{OPD(0xD9, 0xF0), 1, &OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64F2XM1>},
|
||||
{OPD(0xD9, 0xF1), 1, &OpDispatchBuilder::X87FYL2XF64},
|
||||
{OPD(0xD9, 0xF2), 1, &OpDispatchBuilder::X87TANF64},
|
||||
{OPD(0xD9, 0xF3), 1, &OpDispatchBuilder::X87ATANF64},
|
||||
{OPD(0xD9, 0xF4), 1, &OpDispatchBuilder::FXTRACTF64},
|
||||
{OPD(0xD9, 0xF5), 1, &OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM1>},
|
||||
{OPD(0xD9, 0xF6), 1, &OpDispatchBuilder::X87ModifySTP<false>},
|
||||
{OPD(0xD9, 0xF7), 1, &OpDispatchBuilder::X87ModifySTP<true>},
|
||||
{OPD(0xD9, 0xF8), 1, &OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM>},
|
||||
{OPD(0xD9, 0xF9), 1, &OpDispatchBuilder::X87FYL2XF64},
|
||||
{OPD(0xD9, 0xFA), 1, &OpDispatchBuilder::FSQRTF64},
|
||||
{OPD(0xD9, 0xFB), 1, &OpDispatchBuilder::X87SinCosF64},
|
||||
{OPD(0xD9, 0xFC), 1, &OpDispatchBuilder::FRNDINTF64},
|
||||
{OPD(0xD9, 0xFD), 1, &OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64SCALE>},
|
||||
{OPD(0xD9, 0xFE), 1, &OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64SIN>},
|
||||
{OPD(0xD9, 0xFF), 1, &OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64COS>},
|
||||
|
||||
{OPDReg(0xDA, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDA, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDA, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDA, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xDA, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDA, 0xC8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDA, 0xD0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDA, 0xD8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
// E0 = Invalid
|
||||
// E8 = Invalid
|
||||
{OPD(0xDA, 0xE9), 1, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>},
|
||||
// EA = Invalid
|
||||
// F0 = Invalid
|
||||
// F8 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 0) | 0x00, 8, &OpDispatchBuilder::FILDF64},
|
||||
|
||||
{OPDReg(0xDB, 1) | 0x00, 8, &OpDispatchBuilder::FISTF64<true>},
|
||||
|
||||
{OPDReg(0xDB, 2) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
{OPDReg(0xDB, 3) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
// 4 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 5) | 0x00, 8, &OpDispatchBuilder::FLDF64<80>},
|
||||
|
||||
// 6 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::FSTF64<80>},
|
||||
|
||||
|
||||
{OPD(0xDB, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDB, 0xC8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDB, 0xD0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
{OPD(0xDB, 0xD8), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
// E0 = Invalid
|
||||
{OPD(0xDB, 0xE2), 1, &OpDispatchBuilder::NOPOp}, // FNCLEX
|
||||
{OPD(0xDB, 0xE3), 1, &OpDispatchBuilder::FNINITF64},
|
||||
// E4 = Invalid
|
||||
{OPD(0xDB, 0xE8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDB, 0xF0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
|
||||
// F8 = Invalid
|
||||
|
||||
{OPDReg(0xDC, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDC, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDC, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDC, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xDC, 0xC0), 8, &OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xC8), 8, &OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xE0), 8, &OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xE8), 8, &OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xF0), 8, &OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDC, 0xF8), 8, &OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
|
||||
{OPDReg(0xDD, 0) | 0x00, 8, &OpDispatchBuilder::FLDF64<64>},
|
||||
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::FISTF64<true>},
|
||||
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::FSTF64<64>},
|
||||
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::FSTF64<64>},
|
||||
|
||||
{OPDReg(0xDD, 4) | 0x00, 8, &OpDispatchBuilder::X87FRSTORF64},
|
||||
|
||||
// 5 = Invalid
|
||||
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVEF64},
|
||||
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
|
||||
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
|
||||
{OPD(0xDD, 0xD0), 8, &OpDispatchBuilder::FST}, //register-register from regular X87
|
||||
{OPD(0xDD, 0xD8), 8, &OpDispatchBuilder::FST}, //^
|
||||
|
||||
{OPD(0xDD, 0xE0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
{OPD(0xDD, 0xE8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDE, 0) | 0x00, 8, &OpDispatchBuilder::FADDF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 1) | 0x00, 8, &OpDispatchBuilder::FMULF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 2) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDE, 3) | 0x00, 8, &OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>},
|
||||
|
||||
{OPDReg(0xDE, 4) | 0x00, 8, &OpDispatchBuilder::FSUBF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 5) | 0x00, 8, &OpDispatchBuilder::FSUBF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 6) | 0x00, 8, &OpDispatchBuilder::FDIVF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPDReg(0xDE, 7) | 0x00, 8, &OpDispatchBuilder::FDIVF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
{OPD(0xDE, 0xC0), 8, &OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xC8), 8, &OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xD9), 1, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>},
|
||||
{OPD(0xDE, 0xE0), 8, &OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xE8), 8, &OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xF0), 8, &OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
{OPD(0xDE, 0xF8), 8, &OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>},
|
||||
|
||||
{OPDReg(0xDF, 0) | 0x00, 8, &OpDispatchBuilder::FILDF64},
|
||||
|
||||
{OPDReg(0xDF, 1) | 0x00, 8, &OpDispatchBuilder::FISTF64<true>},
|
||||
|
||||
{OPDReg(0xDF, 2) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
{OPDReg(0xDF, 3) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
{OPDReg(0xDF, 4) | 0x00, 8, &OpDispatchBuilder::FBLDF64},
|
||||
|
||||
{OPDReg(0xDF, 5) | 0x00, 8, &OpDispatchBuilder::FILDF64},
|
||||
|
||||
{OPDReg(0xDF, 6) | 0x00, 8, &OpDispatchBuilder::FBSTPF64},
|
||||
|
||||
{OPDReg(0xDF, 7) | 0x00, 8, &OpDispatchBuilder::FISTF64<false>},
|
||||
|
||||
// XXX: This should also set the x87 tag bits to empty
|
||||
// We don't support this currently, so just pop the stack
|
||||
{OPD(0xDF, 0xC0), 8, &OpDispatchBuilder::X87ModifySTP<true>},
|
||||
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPD(0xDF, 0xE8), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDF, 0xF0), 8, &OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> X87OpTable[] = {
|
||||
{OPDReg(0xD8, 0) | 0x00, 8, &OpDispatchBuilder::FADD<32, false, OpDispatchBuilder::OpResult::RES_ST0>},
|
||||
|
||||
@@ -6233,7 +6508,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<4, 8, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<4, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VCMPEQ, 8>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<4>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, false>},
|
||||
@@ -6291,6 +6566,8 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<4>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
@@ -6436,10 +6713,18 @@ constexpr uint16_t PF_F2 = 3;
|
||||
InstallToTable(FEXCore::X86Tables::RepNEModOps, RepNEModOpTable);
|
||||
InstallToTable(FEXCore::X86Tables::OpSizeModOps, OpSizeModOpTable);
|
||||
InstallToTable(FEXCore::X86Tables::SecondInstGroupOps, SecondaryExtensionOpTable);
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
InstallToTable(FEXCore::X86Tables::SecondInstGroupOps, SecondaryExtensionOpTable_64);
|
||||
}
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::SecondModRMTableOps, SecondaryModRMExtensionOpTable);
|
||||
|
||||
InstallToX87Table(FEXCore::X86Tables::X87Ops, X87OpTable);
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
if(ReducedPrecision) {
|
||||
InstallToX87Table(FEXCore::X86Tables::X87Ops, X87F64OpTable);
|
||||
} else {
|
||||
InstallToX87Table(FEXCore::X86Tables::X87Ops, X87OpTable);
|
||||
}
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::H0F38TableOps, H0F38Table);
|
||||
InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3ATable);
|
||||
|
||||
+78
-5
@@ -148,6 +148,7 @@ public:
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::Context *ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
|
||||
void ResetWorkingList();
|
||||
void ResetDecodeFailure() { DecodeFailure = false; }
|
||||
@@ -161,7 +162,9 @@ public:
|
||||
void UnhandledOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void MOVGPROp(OpcodeArgs);
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
void ALUOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
void SyscallOp(OpcodeArgs);
|
||||
@@ -463,6 +466,57 @@ public:
|
||||
template<size_t width, bool Integer, FCOMIFlags whichflags, bool poptwice>
|
||||
void FCOMI(OpcodeArgs);
|
||||
|
||||
// F64 X87 Ops
|
||||
template<size_t width>
|
||||
void FLDF64(OpcodeArgs);
|
||||
template<uint64_t num>
|
||||
void FLDF64_Const(OpcodeArgs);
|
||||
|
||||
void FBLDF64(OpcodeArgs);
|
||||
void FBSTPF64(OpcodeArgs);
|
||||
|
||||
void FILDF64(OpcodeArgs);
|
||||
|
||||
template<size_t width>
|
||||
void FSTF64(OpcodeArgs);
|
||||
|
||||
void FSTF64(OpcodeArgs);
|
||||
|
||||
template<bool Truncate>
|
||||
void FISTF64(OpcodeArgs);
|
||||
|
||||
template<size_t width, bool Integer, OpResult ResInST0>
|
||||
void FADDF64(OpcodeArgs);
|
||||
template<size_t width, bool Integer, OpResult ResInST0>
|
||||
void FMULF64(OpcodeArgs);
|
||||
template<size_t width, bool Integer, bool reverse, OpResult ResInST0>
|
||||
void FDIVF64(OpcodeArgs);
|
||||
template<size_t width, bool Integer, bool reverse, OpResult ResInST0>
|
||||
void FSUBF64(OpcodeArgs);
|
||||
void FCHSF64(OpcodeArgs);
|
||||
void FABSF64(OpcodeArgs);
|
||||
void FTSTF64(OpcodeArgs);
|
||||
void FRNDINTF64(OpcodeArgs);
|
||||
void FXTRACTF64(OpcodeArgs);
|
||||
void FNINITF64(OpcodeArgs);
|
||||
void FSQRTF64(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp>
|
||||
void X87UnaryOpF64(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp>
|
||||
void X87BinaryOpF64(OpcodeArgs);
|
||||
void X87SinCosF64(OpcodeArgs);
|
||||
void X87FLDCWF64(OpcodeArgs);
|
||||
void X87FYL2XF64(OpcodeArgs);
|
||||
void X87TANF64(OpcodeArgs);
|
||||
void X87ATANF64(OpcodeArgs);
|
||||
void X87FNSAVEF64(OpcodeArgs);
|
||||
void X87FRSTORF64(OpcodeArgs);
|
||||
void X87FXAMF64(OpcodeArgs);
|
||||
void X87LDENVF64(OpcodeArgs);
|
||||
|
||||
template<size_t width, bool Integer, FCOMIFlags whichflags, bool poptwice>
|
||||
void FCOMIF64(OpcodeArgs);
|
||||
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
@@ -535,6 +589,15 @@ public:
|
||||
|
||||
void PSADBW(OpcodeArgs);
|
||||
|
||||
void SHA1NEXTEOp(OpcodeArgs);
|
||||
void SHA1MSG1Op(OpcodeArgs);
|
||||
void SHA1MSG2Op(OpcodeArgs);
|
||||
void SHA1RNDS4Op(OpcodeArgs);
|
||||
|
||||
void SHA256MSG1Op(OpcodeArgs);
|
||||
void SHA256MSG2Op(OpcodeArgs);
|
||||
void SHA256RNDS2Op(OpcodeArgs);
|
||||
|
||||
void AESImcOp(OpcodeArgs);
|
||||
void AESEncOp(OpcodeArgs);
|
||||
void AESEncLastOp(OpcodeArgs);
|
||||
@@ -580,12 +643,22 @@ private:
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
ACCESS_DEFAULT,
|
||||
// TSO access behaviour
|
||||
ACCESS_TSO,
|
||||
// Non-TSO access behaviour
|
||||
ACCESS_NONTSO,
|
||||
// Non-temporal streaming
|
||||
ACCESS_STREAM,
|
||||
};
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
|
||||
@@ -10,13 +10,259 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <stdint.h>
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto Tmp = _Ror(_VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
|
||||
auto Top = _Add(_VExtractToGPR(16, 4, Src, 3), Tmp);
|
||||
auto Result = _VInsGPR(16, 4, 3, Src, Top);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W5 = _VExtractToGPR(16, 4, Src, 2);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, _Xor(W2, W0));
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, _Xor(W3, W1));
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, _Xor(W4, W2));
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, _Xor(W5, W3));
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
// ROR by 31 is equivalent to a ROL by 1
|
||||
auto ThirtyOne = _Constant(32, 31);
|
||||
|
||||
auto W13 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W16 = _Ror(_Xor(_VExtractToGPR(16, 4, Dest, 3), W13), ThirtyOne);
|
||||
auto W17 = _Ror(_Xor(_VExtractToGPR(16, 4, Dest, 2), W14), ThirtyOne);
|
||||
auto W18 = _Ror(_Xor(_VExtractToGPR(16, 4, Dest, 1), W15), ThirtyOne);
|
||||
auto W19 = _Ror(_Xor(_VExtractToGPR(16, 4, Dest, 0), W16), ThirtyOne);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W16);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W17);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W18);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W19);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(),
|
||||
"Src1 needs to be literal here to indicate function and constants");
|
||||
|
||||
using FnType = OrderedNode* (*)(OpDispatchBuilder&, OrderedNode*, OrderedNode*, OrderedNode*);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(Self._And(B, C), Self._And(Self._Not(B), D));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(Self._Xor(B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(Self._Xor(Self._And(B, C), Self._And(B, D)), Self._And(C, D));
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
return Self._Xor(Self._Xor(B, C), D);
|
||||
};
|
||||
|
||||
constexpr std::array<uint32_t, 4> k_array{
|
||||
0x5A827999U,
|
||||
0x6ED9EBA1U,
|
||||
0x8F1BBCDCU,
|
||||
0xCA62C1D6U,
|
||||
};
|
||||
|
||||
constexpr std::array<FnType, 4> fn_array{
|
||||
f0, f1, f2, f3,
|
||||
};
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Data.Literal.Value & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W1 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W2 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto W3 = _VExtractToGPR(16, 4, Src, 0);
|
||||
|
||||
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto B = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto C = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto D = _VExtractToGPR(16, 4, Dest, 0);
|
||||
|
||||
auto A1 = _Add(_Add(_Add(Fn(*this, B, C, D), _Ror(A, _Constant(32, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(B, _Constant(32, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C,
|
||||
OrderedNode *D, OrderedNode *E, OrderedNode *W) -> RoundResult {
|
||||
auto ANext = _Add(_Add(_Add(_Add(Fn(*this, B, C, D), _Ror(A, _Constant(32, 27))), W), E), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(B, _Constant(32, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, W1);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, W2);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, W3);
|
||||
|
||||
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(16, 4, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(16, 4, 0, Dest1, std::get<3>(Final));
|
||||
|
||||
StoreResult(FPRClass, Op, Dest0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(_Xor(_Ror(W, _Constant(32, 7)), _Ror(W, _Constant(32, 18))), _Lshr(W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
|
||||
auto Sig3 = _Add(W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(W0, Sigma0(W1));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
const auto Sigma1 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(_Xor(_Ror(W, _Constant(32, 17)), _Ror(W, _Constant(32, 19))), _Lshr(W, _Constant(32, 10)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W16 = _Add(_VExtractToGPR(16, 4, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(_VExtractToGPR(16, 4, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(_VExtractToGPR(16, 4, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(_VExtractToGPR(16, 4, Dest, 3), Sigma1(W17));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W16);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
|
||||
return _Xor(_And(E, F), _And(_Not(E), G));
|
||||
};
|
||||
const auto Major = [this](OrderedNode *A, OrderedNode *B, OrderedNode *C) -> OrderedNode* {
|
||||
return _Xor(_Xor(_And(A, B), _And(A, C)), _And(B, C));
|
||||
};
|
||||
const auto Sigma0 = [this](OrderedNode *A) -> OrderedNode* {
|
||||
return _Xor(_Xor(_Ror(A, _Constant(32, 2)), _Ror(A, _Constant(32, 13))), _Ror(A, _Constant(32, 22)));
|
||||
};
|
||||
const auto Sigma1 = [this](OrderedNode *E) -> OrderedNode* {
|
||||
return _Xor(_Xor(_Ror(E, _Constant(32, 6)), _Ror(E, _Constant(32, 11))), _Ror(E, _Constant(32, 25)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *XMM0 = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[0]));
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
|
||||
|
||||
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*,
|
||||
OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
|
||||
const auto Round = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C, OrderedNode *D,
|
||||
OrderedNode *E, OrderedNode *F, OrderedNode *G, OrderedNode *H,
|
||||
OrderedNode* WK) -> RoundResult {
|
||||
auto ANext = _Add(_Add(_Add(_Add(_Add(Ch(E, F, G), Sigma1(E)), WK), H), Major(A, B, C)), Sigma0(A));
|
||||
auto BNext = A;
|
||||
auto CNext = B;
|
||||
auto DNext = C;
|
||||
auto ENext = _Add(_Add(_Add(_Add(Ch(E, F, G), Sigma1(E)), WK), H), D);
|
||||
auto FNext = E;
|
||||
auto GNext = F;
|
||||
auto HNext = G;
|
||||
|
||||
return {ANext, BNext, CNext, DNext, ENext, FNext, GNext, HNext};
|
||||
};
|
||||
|
||||
|
||||
auto [A1, B1, C1, D1, E1, F1, G1, H1] = Round(A0, B0, C0, D0, E0, F0, G0, H0, WK0);
|
||||
auto Final = Round(A1, B1, C1, D1, E1, F1, G1, H1, WK1);
|
||||
|
||||
auto Res3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Res2 = _VInsGPR(16, 4, 2, Res3, std::get<1>(Final));
|
||||
auto Res1 = _VInsGPR(16, 4, 1, Res2, std::get<4>(Final));
|
||||
auto Res0 = _VInsGPR(16, 4, 0, Res1, std::get<5>(Final));
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Res = _VAESImc(Src);
|
||||
|
||||
@@ -28,6 +28,11 @@ void OpDispatchBuilder::MOVVectorOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1, true, false, MemoryAccessType::ACCESS_STREAM);
|
||||
StoreResult(FPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVAPSOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
@@ -750,7 +755,7 @@ void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
}
|
||||
else {
|
||||
// If we are storing to memory then we store the size of the element extracted
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, ElementSize, -1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1194,7 +1199,8 @@ void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
{
|
||||
auto DestByte = _Bfe(8, 8 * Select, DestElement);
|
||||
auto MemLocation = _Add(MemDest, _Constant(Element * 8 + Select));
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemLocation, DestByte, 1);
|
||||
// MASKMOVDQU/MASKMOVQ is explicitly weakly-ordered on its store
|
||||
_StoreMem(GPRClass, 1, MemLocation, DestByte, 1);
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
@@ -1809,7 +1815,6 @@ void OpDispatchBuilder::VPFCMPOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
ShouldDump = true;
|
||||
}
|
||||
|
||||
template
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -167,7 +167,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC2, 1, X86InstInfo{"RET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2, nullptr}},
|
||||
{0xC3, 1, X86InstInfo{"RET", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0, nullptr}},
|
||||
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3, nullptr}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0, nullptr}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_DEBUG , 1, nullptr}},
|
||||
|
||||
@@ -87,6 +87,14 @@ void InitializeH0F38Tables() {
|
||||
{OPD(PF_38_66, 0x40), 1, X86InstInfo{"PMULLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x41), 1, X86InstInfo{"PHMINPOSUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_NONE, 0xC8), 1, X86InstInfo{"SHA1NEXTE", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, X86InstInfo{"SHA1MSG1", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, X86InstInfo{"SHA1MSG2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_NONE, 0xCB), 1, X86InstInfo{"SHA256RNDS2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, X86InstInfo{"SHA256MSG1", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, X86InstInfo{"SHA256MSG2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, X86InstInfo{"AESIMC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDC), 1, X86InstInfo{"AESENC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDD), 1, X86InstInfo{"AESENCLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
@@ -31,7 +31,7 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_8BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
@@ -49,6 +49,8 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
};
|
||||
|
||||
|
||||
@@ -343,8 +343,8 @@ void InitializeSecondaryGroupTables() {
|
||||
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 0), 1, X86InstInfo{"RDFSBASE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 1), 1, X86InstInfo{"RDGSBASE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 2), 1, X86InstInfo{"WRFSBASE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 3), 1, X86InstInfo{"WRGSBASE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 2), 1, X86InstInfo{"WRFSBASE", TYPE_INST, GenFlagsDstSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 3), 1, X86InstInfo{"WRGSBASE", TYPE_INST, GenFlagsDstSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 5), 1, X86InstInfo{"INCSSPQ", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_F3, 6), 1, X86InstInfo{"CLRSSBSY", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
+63
-74
@@ -4,6 +4,8 @@
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <Interface/Core/LookupCache.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -77,7 +79,7 @@ namespace FEXCore::IR {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool LoadAOTIRCache(AOTCacheType *AOTIRCache, int streamfd) {
|
||||
static bool LoadAOTIRCache(AOTIRCacheEntry *Entry, int streamfd) {
|
||||
uint64_t tag;
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != FEXCore::IR::AOTIR_COOKIE)
|
||||
@@ -99,6 +101,10 @@ namespace FEXCore::IR {
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size()))
|
||||
return false;
|
||||
|
||||
if (Entry->FileId != Module) {
|
||||
return false;
|
||||
}
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize)))
|
||||
@@ -119,19 +125,16 @@ namespace FEXCore::IR {
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
AOTIRCache->insert({Module, {Array, FilePtr, Size}});
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
Entry->Array = Array;
|
||||
Entry->FilePtr = FilePtr;
|
||||
Entry->Size = Size;
|
||||
|
||||
LogMan::Msg::DFmt("AOTIR: Module {} has {} functions", Module, Array->Count);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::~AOTIRCaptureCache() {
|
||||
for (auto &Mod: AOTIRCache) {
|
||||
FEXCore::Allocator::munmap(Mod.second.mapping, Mod.second.size);
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
|
||||
@@ -230,39 +233,27 @@ namespace FEXCore::IR {
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &File: FilesWithCode) {
|
||||
Writer(File.first, File.second);
|
||||
for( const auto &Entry: AOTIRCache) {
|
||||
if (Entry.second.ContainsCode) {
|
||||
Writer(Entry.second.FileId, Entry.second.Filename);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
{
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (!file->second.ContainsCode) {
|
||||
file->second.ContainsCode = true;
|
||||
FilesWithCode[file->second.fileid] = file->second.filename;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
PreGenerateIRFetchResult Result{};
|
||||
if (IRList == nullptr && CTX->Config.AOTIRLoad()) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
auto Mod = (FEXCore::IR::AOTIRInlineIndex*)file->second.CachedFileEntry;
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
AOTIRCacheEntry.Entry->ContainsCode = true;
|
||||
|
||||
if (Mod == nullptr) {
|
||||
file->second.CachedFileEntry = Mod = AOTIRCache[file->second.fileid].Array;
|
||||
}
|
||||
if (IRList == nullptr && CTX->Config.AOTIRLoad()) {
|
||||
auto Mod = AOTIRCacheEntry.Entry->Array;
|
||||
|
||||
if (Mod != nullptr)
|
||||
{
|
||||
auto AOTEntry = Mod->Find(GuestRIP - file->second.Start + file->second.Offset);
|
||||
auto AOTEntry = Mod->Find(GuestRIP - AOTIRCacheEntry.Offset);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
@@ -299,18 +290,15 @@ namespace FEXCore::IR {
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR,
|
||||
bool DecrementRefCount) {
|
||||
bool GeneratedIR) {
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming()) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto file = FindAddrForFile(StartAddr, Length);
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
// Only go down this path if we actually found a library region
|
||||
if (file != AddrToFile.end()) {
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
CTX->Symbols.RegisterNamedRegion(CodePtr, DebugData->HostCodeSize, file->second.filename);
|
||||
CTX->Symbols.RegisterNamedRegion(CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
@@ -319,29 +307,31 @@ namespace FEXCore::IR {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - file->second.Start + file->second.Offset;
|
||||
auto LocalStartAddr = StartAddr - file->second.Start + file->second.Offset;
|
||||
auto fileid = file->second.fileid;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRList, RAData, fileid]() {
|
||||
auto *AotFile = &AOTIRCaptureCacheMap[fileid];
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.Offset;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.Offset;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
auto RADataCopy = RAData->CreateCopy();
|
||||
auto IRListCopy = IRList->CreateCopy();
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy, FileId]() {
|
||||
|
||||
// It is guaranteed via AOTIRCaptureCacheWriteoutLock and AOTIRCaptureCacheWriteoutFlusing that this will not run concurrently
|
||||
// Memory coherency is guaranteed via AOTIRCaptureCacheWriteoutLock
|
||||
|
||||
auto *AotFile = &AOTIRCaptureCacheMap[FileId];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(fileid);
|
||||
AotFile->Stream = AOTIRWriter(FileId);
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->write((char*)&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRList, RAData);
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy);
|
||||
FEXCore::Allocator::free(RADataCopy);
|
||||
delete IRListCopy;
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
|
||||
if (DecrementRefCount) {
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
}
|
||||
|
||||
Thread->CPUBackend->ClearCache();
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -352,6 +342,8 @@ namespace FEXCore::IR {
|
||||
if (Thread->CPUBackend->NeedsRetainedIRCopy()) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
else {
|
||||
@@ -366,21 +358,7 @@ namespace FEXCore::IR {
|
||||
return false;
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::AddrToFileMapType::iterator AOTIRCaptureCache::FindAddrForFile(uint64_t Entry, uint64_t Length) {
|
||||
// Thread safety here! We are returning an iterator to the map object
|
||||
// This needs the AOTIRCacheLock locked prior to coming in to the function
|
||||
auto file = AddrToFile.lower_bound(Entry);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (file->second.Start <= Entry && (file->second.Start + file->second.Len) >= (Entry + Length)) {
|
||||
return file;
|
||||
}
|
||||
}
|
||||
return AddrToFile.end();
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// TODO: Support overlapping maps and region splitting
|
||||
AOTIRCacheEntry *AOTIRCaptureCache::LoadAOTIRCacheEntry(const std::string &filename) {
|
||||
auto base_filename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!base_filename.empty()) {
|
||||
@@ -396,21 +374,32 @@ namespace FEXCore::IR {
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
AddrToFile.insert({ Base, { Base, Size, Offset, fileid, filename, nullptr, false} });
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry{0, 0, 0, fileid, filename, false}});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
if (CTX->Config.AOTIRLoad && !AOTIRCache.contains(fileid) && AOTIRLoader) {
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
|
||||
if (CTX->Config.AOTIRLoad && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
if (streamfd != -1) {
|
||||
FEXCore::IR::LoadAOTIRCache(&AOTIRCache, streamfd);
|
||||
FEXCore::IR::LoadAOTIRCache(Entry, streamfd);
|
||||
close(streamfd);
|
||||
}
|
||||
}
|
||||
return Entry;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
// TODO: Support partial removing
|
||||
AddrToFile.erase(Base);
|
||||
void AOTIRCaptureCache::UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry) {
|
||||
LOGMAN_THROW_A_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
|
||||
if (Entry->Array) {
|
||||
FEXCore::Allocator::munmap(Entry->FilePtr, Entry->Size);
|
||||
Entry->Array = nullptr;
|
||||
Entry->FilePtr = nullptr;
|
||||
Entry->Size = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
+8
-24
@@ -72,18 +72,19 @@ namespace FEXCore::IR {
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex *Array;
|
||||
void *mapping;
|
||||
size_t size;
|
||||
void *FilePtr;
|
||||
size_t Size;
|
||||
std::string FileId;
|
||||
std::string Filename;
|
||||
bool ContainsCode;
|
||||
};
|
||||
|
||||
using AOTCacheType = std::unordered_map<std::string, FEXCore::IR::AOTIRCacheEntry>;
|
||||
bool LoadAOTIRCache(AOTCacheType *AOTIRCache, int streamfd);
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::Context *ctx) : CTX {ctx} {}
|
||||
~AOTIRCaptureCache();
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
@@ -108,11 +109,10 @@ namespace FEXCore::IR {
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR,
|
||||
bool DecrementRefCount);
|
||||
bool GeneratedIR);
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry);
|
||||
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
|
||||
@@ -136,27 +136,11 @@ namespace FEXCore::IR {
|
||||
|
||||
std::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
std::map<std::string, std::string> FilesWithCode;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
uint64_t Len;
|
||||
uint64_t Offset;
|
||||
std::string fileid;
|
||||
std::string filename;
|
||||
void *CachedFileEntry;
|
||||
bool ContainsCode;
|
||||
};
|
||||
|
||||
using AddrToFileMapType = std::map<uint64_t, AddrToFileEntry>;
|
||||
AddrToFileMapType AddrToFile;
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
std::function<int(const std::string&)> AOTIRLoader;
|
||||
std::function<std::unique_ptr<std::ofstream>(const std::string&)> AOTIRWriter;
|
||||
std::function<void(const std::string&)> AOTIRRenamer;
|
||||
std::unordered_map<std::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
|
||||
AddrToFileMapType::iterator FindAddrForFile(uint64_t Entry, uint64_t Length);
|
||||
};
|
||||
}
|
||||
+35
-1
@@ -187,7 +187,7 @@
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"RemoveCodeEntry": {
|
||||
"RemoveThreadCodeEntry": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
@@ -231,6 +231,11 @@
|
||||
],
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
"Yield": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["This is a hint instruction that the CPU is likely to do a spin so it might want to pause to help out SMP",
|
||||
"Can be implemented as a NOP if necessary"]
|
||||
}
|
||||
},
|
||||
"Branch": {
|
||||
@@ -1478,6 +1483,35 @@
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
}
|
||||
},
|
||||
"F64": {
|
||||
"FPR = F64ATAN FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64F2XM1 FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FYL2X FPR:$Src, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64SIN FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
}
|
||||
},
|
||||
"F80": {
|
||||
"F80LoadFCW GPR:$Src": {
|
||||
"HasSideEffects": true
|
||||
|
||||
+4
-3
@@ -309,7 +309,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
LineDefinition *CurrentDef{};
|
||||
std::unordered_map<std::string_view, FEXCore::IR::IROps> NameToOpMap;
|
||||
|
||||
IRParser(std::istream *text) {
|
||||
IRParser(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, std::istream *text)
|
||||
: IREmitter {ThreadAllocator} {
|
||||
InitializeNameMap();
|
||||
|
||||
std::string TmpLine;
|
||||
@@ -660,8 +661,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
|
||||
} // anon namespace
|
||||
|
||||
std::unique_ptr<IREmitter> Parse(std::istream *in) {
|
||||
auto parser = std::make_unique<IRParser>(in);
|
||||
std::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, std::istream *in) {
|
||||
auto parser = std::make_unique<IRParser>(ThreadAllocator, in);
|
||||
|
||||
if (parser->Loaded) {
|
||||
return parser;
|
||||
|
||||
+4
-3
@@ -6,6 +6,7 @@ desc: Defines which passes are run, and runs them
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -15,7 +16,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
|
||||
void PassManager::AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllocation) {
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineConstants, bool StaticRegisterAllocation) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
@@ -29,7 +30,7 @@ void PassManager::AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllo
|
||||
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
InsertPass(CreateConstProp(InlineConstants));
|
||||
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9));
|
||||
|
||||
////// InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
|
||||
@@ -48,7 +49,7 @@ void PassManager::AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllo
|
||||
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
// Compact before IR, don't worry about RA generating spills/fills
|
||||
InsertPass(CreateIRCompaction(), "Compaction");
|
||||
InsertPass(CreateIRCompaction(ctx->OpDispatcherAllocator), "Compaction");
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultValidationPasses() {
|
||||
|
||||
+2
-1
@@ -7,6 +7,7 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
@@ -39,7 +40,7 @@ protected:
|
||||
class PassManager final {
|
||||
friend class SyscallOptimization;
|
||||
public:
|
||||
void AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllocation);
|
||||
void AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineConstants, bool StaticRegisterAllocation);
|
||||
void AddDefaultValidationPasses();
|
||||
Pass* InsertPass(std::unique_ptr<Pass> Pass, std::string Name = "") {
|
||||
Pass->RegisterPassManager(this);
|
||||
|
||||
+6
-2
@@ -2,18 +2,22 @@
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class IntrusivePooledAllocator;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateSyscallOptimization();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
+78
-39
@@ -65,11 +65,20 @@ static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
|
||||
static bool IsImmMemory(uint64_t imm, uint8_t AccessSize) {
|
||||
if ( ((int64_t)imm >= -255) && ((int64_t)imm <= 256) )
|
||||
return true;
|
||||
return true;
|
||||
else if ( (imm & (AccessSize-1)) == 0 && imm/AccessSize <= 4095 )
|
||||
return true;
|
||||
return true;
|
||||
else {
|
||||
return false;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
static bool IsTSOImm9(uint64_t imm) {
|
||||
// RCPC2 only has a 9-bit signed offset
|
||||
if ( ((int64_t)imm >= -256) && ((int64_t)imm <= 255) )
|
||||
return true;
|
||||
else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -128,30 +137,30 @@ static std::tuple<MemOffsetType, uint8_t, OrderedNode*, OrderedNode*> MemExtende
|
||||
}
|
||||
|
||||
static OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t mask) {
|
||||
#if 1 // HOTFIX: We need to clear up the meaning of opsize and dest size. See #594
|
||||
return src;
|
||||
#else
|
||||
auto IROp = IREmit->GetOpHeader(src);
|
||||
if (IROp->Op == OP_AND) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
uint64_t imm;
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &imm) && ((imm & mask) == mask)) {
|
||||
#if 1 // HOTFIX: We need to clear up the meaning of opsize and dest size. See #594
|
||||
return src;
|
||||
#else
|
||||
auto IROp = IREmit->GetOpHeader(src);
|
||||
if (IROp->Op == OP_AND) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
uint64_t imm;
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &imm) && ((imm & mask) == mask)) {
|
||||
return RemoveUselessMasking(IREmit, IROp->Args[0], mask);
|
||||
}
|
||||
} else if (IROp->Op == OP_BFE) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
if (Op->lsb == 0) {
|
||||
uint64_t imm = 1ULL << (Op->Width-1);
|
||||
imm = (imm-1) *2 + 1;
|
||||
|
||||
if ((imm & mask) == mask) {
|
||||
return RemoveUselessMasking(IREmit, IROp->Args[0], mask);
|
||||
}
|
||||
} else if (IROp->Op == OP_BFE) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
if (Op->lsb == 0) {
|
||||
uint64_t imm = 1ULL << (Op->Width-1);
|
||||
imm = (imm-1) *2 + 1;
|
||||
|
||||
if ((imm & mask) == mask) {
|
||||
return RemoveUselessMasking(IREmit, IROp->Args[0], mask);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return src;
|
||||
#endif
|
||||
return src;
|
||||
#endif
|
||||
}
|
||||
|
||||
static bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t Width) {
|
||||
@@ -167,7 +176,9 @@ static bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool DoInlineConstants) : InlineConstants(DoInlineConstants) { }
|
||||
explicit ConstProp(bool DoInlineConstants, bool SupportsTSOImm9)
|
||||
: InlineConstants(DoInlineConstants)
|
||||
, SupportsTSOImm9 {SupportsTSOImm9} { }
|
||||
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
|
||||
@@ -179,13 +190,14 @@ private:
|
||||
void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
bool ConstantPropagation(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
bool ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
|
||||
std::unordered_map<uint64_t, OrderedNode*> ConstPool;
|
||||
std::map<OrderedNode*, uint64_t> AddressgenConsts;
|
||||
bool SupportsTSOImm9{};
|
||||
};
|
||||
|
||||
bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
@@ -229,8 +241,8 @@ void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& Cur
|
||||
// the value isn't used after the select otherwise
|
||||
// make sure the sizes match
|
||||
if (SelectOpHdr->Size == UnaryOpHdr->Size && SelectOpHdr->Op == OP_SELECT && SelectOpNode->NumUses == 1
|
||||
&& IREmit->IsValueConstant(SelectOp->TrueVal)
|
||||
&& IREmit->IsValueConstant(SelectOp->FalseVal)) {
|
||||
&& IREmit->IsValueConstant(SelectOp->TrueVal)
|
||||
&& IREmit->IsValueConstant(SelectOp->FalseVal)) {
|
||||
|
||||
IREmit->SetWriteCursor(IREmit->UnwrapNode(SelectOpNode->Header.Previous));
|
||||
|
||||
@@ -526,8 +538,7 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
case OP_FINDLSB:
|
||||
case OP_FINDMSB:
|
||||
case OP_REV:
|
||||
case OP_SBFE:
|
||||
{
|
||||
case OP_SBFE: {
|
||||
uint64_t Constant1;
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
@@ -835,7 +846,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
{
|
||||
@@ -853,7 +863,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
@@ -884,7 +893,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CONDJUMP:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
@@ -901,7 +909,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_EXITFUNCTION:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
@@ -926,7 +933,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_AND:
|
||||
@@ -945,7 +951,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
@@ -962,7 +967,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STOREMEM:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
@@ -979,7 +983,42 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (SupportsTSOImm9) {
|
||||
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) {
|
||||
if (IsTSOImm9(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (SupportsTSOImm9) {
|
||||
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) {
|
||||
if (IsTSOImm9(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1018,8 +1057,8 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
return Changed;
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants) {
|
||||
return std::make_unique<ConstProp>(InlineConstants);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9) {
|
||||
return std::make_unique<ConstProp>(InlineConstants, SupportsTSOImm9);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -31,7 +31,7 @@ static_assert(sizeof(RemapNode) == 4);
|
||||
|
||||
class IRCompaction final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
IRCompaction();
|
||||
IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
|
||||
private:
|
||||
@@ -46,12 +46,14 @@ private:
|
||||
std::vector<CodeBlockData> GeneratedCodeBlocks{};
|
||||
};
|
||||
|
||||
IRCompaction::IRCompaction()
|
||||
: LocalBuilder {nullptr} {
|
||||
IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
: LocalBuilder {Allocator} {
|
||||
OldToNewRemap.resize(AlignSize);
|
||||
}
|
||||
|
||||
bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
LocalBuilder.ReownOrClaimBuffer();
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
uint32_t NodeCount = CurrentIR.GetSSACount();
|
||||
|
||||
@@ -211,11 +213,12 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
|
||||
IREmit->CopyData(LocalBuilder);
|
||||
|
||||
LocalBuilder.DelayedDisownBuffer();
|
||||
return true;
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction() {
|
||||
return std::make_unique<IRCompaction>();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator) {
|
||||
return std::make_unique<IRCompaction>(Allocator);
|
||||
}
|
||||
|
||||
}
|
||||
+20
-2
@@ -27,10 +27,21 @@ namespace Handler {
|
||||
static inline std::string_view SMCCheckHandler(std::string_view Value) {
|
||||
if (Value == "none")
|
||||
return "0";
|
||||
else if (Value == "mman")
|
||||
else if (Value == "mtrack")
|
||||
return "1";
|
||||
else if (Value == "full")
|
||||
return "2";
|
||||
else if (Value == "mman")
|
||||
return "3";
|
||||
return "0";
|
||||
}
|
||||
static inline std::string_view CacheObjectCodeHandler(std::string_view Value) {
|
||||
if (Value == "none")
|
||||
return "0";
|
||||
else if (Value == "read")
|
||||
return "1";
|
||||
else if (Value == "write")
|
||||
return "2";
|
||||
return "0";
|
||||
}
|
||||
}
|
||||
@@ -48,8 +59,15 @@ namespace Handler {
|
||||
|
||||
enum ConfigSMCChecks {
|
||||
CONFIG_SMC_NONE,
|
||||
CONFIG_SMC_MMAN,
|
||||
CONFIG_SMC_MTRACK,
|
||||
CONFIG_SMC_FULL,
|
||||
CONFIG_SMC_MMAN,
|
||||
};
|
||||
|
||||
enum ConfigObjectCodeHandler {
|
||||
CONFIG_NONE,
|
||||
CONFIG_READ,
|
||||
CONFIG_READWRITE,
|
||||
};
|
||||
|
||||
enum class LayerType {
|
||||
|
||||
+20
-2
@@ -25,6 +25,10 @@ namespace Core {
|
||||
struct CpuStateFrame;
|
||||
}
|
||||
|
||||
namespace CodeSerialize {
|
||||
struct CodeObjectFileSection;
|
||||
}
|
||||
|
||||
namespace CPU {
|
||||
class InterpreterCore;
|
||||
class JITCore;
|
||||
@@ -59,6 +63,16 @@ class LLVMCore;
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
*
|
||||
* @param Entry - RIP of the entry
|
||||
* @param SerializationData - Serialization data referring to the object cache for `Entry`
|
||||
*
|
||||
* @return An executable function pointer relocated from the cache object
|
||||
*/
|
||||
[[nodiscard]] virtual void *RelocateJITObjectCode(uint64_t Entry, CodeSerialize::CodeObjectFileSection const *SerializationData) { return nullptr; }
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
*
|
||||
@@ -89,8 +103,7 @@ class LLVMCore;
|
||||
}
|
||||
|
||||
virtual void ClearCache() {}
|
||||
virtual void CopyNecessaryDataForCompileThread(CPUBackend *Original) {}
|
||||
virtual bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const { return false; }
|
||||
virtual bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const { return false; }
|
||||
|
||||
/**
|
||||
* @brief Does this CPUBackend need its IR to stick around for correct emulation
|
||||
@@ -99,6 +112,11 @@ class LLVMCore;
|
||||
*/
|
||||
virtual bool NeedsRetainedIRCopy() const { return false; }
|
||||
|
||||
/**
|
||||
* @brief Clear any relocations after JIT compiling
|
||||
*/
|
||||
virtual void ClearRelocations() {}
|
||||
|
||||
using AsmDispatch = FEX_NAKED void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = FEX_NAKED void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
|
||||
+8
-3
@@ -31,6 +31,10 @@ namespace FEXCore::HLE {
|
||||
class SyscallHandler;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct AOTIRCacheEntry;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
enum ExitReason {
|
||||
@@ -236,15 +240,16 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf);
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name);
|
||||
FEX_DEFAULT_VISIBILITY void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length);
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string& Name);
|
||||
FEX_DEFAULT_VISIBILITY void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, FEXCore::IR::AOTIRCacheEntry *Entry);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void FinalizeAOTIRCache(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
FEX_DEFAULT_VISIBILITY void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length);
|
||||
FEX_DEFAULT_VISIBILITY void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress);
|
||||
}
|
||||
+12
-2
@@ -78,6 +78,16 @@ namespace FEXCore::Core {
|
||||
OPINDEX_F80FPREM,
|
||||
OPINDEX_F80SCALE,
|
||||
|
||||
// Double Precision
|
||||
OPINDEX_F64SIN,
|
||||
OPINDEX_F64COS,
|
||||
OPINDEX_F64TAN,
|
||||
OPINDEX_F64ATAN,
|
||||
OPINDEX_F64F2XM1,
|
||||
OPINDEX_F64FYL2X,
|
||||
OPINDEX_F64FPREM,
|
||||
OPINDEX_F64FPREM1,
|
||||
OPINDEX_F64SCALE,
|
||||
// Maximum
|
||||
OPINDEX_MAX,
|
||||
};
|
||||
@@ -91,7 +101,7 @@ namespace FEXCore::Core {
|
||||
uint64_t LREM{};
|
||||
uint64_t PrintValue{};
|
||||
uint64_t PrintVectorValue{};
|
||||
uint64_t RemoveCodeEntryFromJIT{};
|
||||
uint64_t RemoveThreadCodeEntryFromJIT{};
|
||||
uint64_t CPUIDObj{};
|
||||
uint64_t CPUIDFunction{};
|
||||
uint64_t SyscallHandlerObj{};
|
||||
@@ -124,7 +134,7 @@ namespace FEXCore::Core {
|
||||
// Process specific
|
||||
uint64_t PrintValue{};
|
||||
uint64_t PrintVectorValue{};
|
||||
uint64_t RemoveCodeEntryFromJIT{};
|
||||
uint64_t RemoveThreadCodeEntryFromJIT{};
|
||||
uint64_t CPUIDObj{};
|
||||
uint64_t CPUIDFunction{};
|
||||
uint64_t SyscallHandlerObj{};
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <unordered_map>
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
class LookupCache;
|
||||
@@ -19,6 +20,10 @@ namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
class Decoder;
|
||||
}
|
||||
@@ -48,6 +53,7 @@ namespace FEXCore::Core {
|
||||
struct DebugData {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
std::vector<DebugDataSubblock> Subblocks;
|
||||
std::vector<FEXCore::CPU::Relocation> *Relocations;
|
||||
};
|
||||
|
||||
enum class SignalEvent {
|
||||
@@ -97,9 +103,9 @@ namespace FEXCore::Core {
|
||||
|
||||
int StatusCode{};
|
||||
FEXCore::Context::ExitReason ExitReason {FEXCore::Context::ExitReason::EXIT_WAITING};
|
||||
uint32_t CompileBlockReentrantRefCount{};
|
||||
std::shared_ptr<FEXCore::CompileService> CompileService;
|
||||
bool IsCompileService{false};
|
||||
|
||||
std::shared_mutex ObjectCacheRefCounter{};
|
||||
bool DestroyedByParent{false}; // Should the parent destroy this thread, or it destory itself
|
||||
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
|
||||
|
||||
@@ -1,13 +1,19 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <shared_mutex>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXHeaderUtils/ScopedSignalMask.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct AOTIRCacheEntry;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -43,6 +49,24 @@ namespace FEXCore::HLE {
|
||||
OS_HANGOVER,
|
||||
};
|
||||
|
||||
class SyscallHandler;
|
||||
struct AOTIRCacheEntryLookupResult {
|
||||
AOTIRCacheEntryLookupResult(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t Offset, FHU::ScopedSignalMaskWithSharedLock &&lk)
|
||||
: Entry(Entry), Offset(Offset), lk(std::move(lk))
|
||||
{
|
||||
|
||||
}
|
||||
|
||||
AOTIRCacheEntryLookupResult(AOTIRCacheEntryLookupResult&&) = default;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry *Entry;
|
||||
uintptr_t Offset;
|
||||
|
||||
friend class SyscallHandler;
|
||||
protected:
|
||||
FHU::ScopedSignalMaskWithSharedLock lk;
|
||||
};
|
||||
|
||||
class SyscallHandler {
|
||||
public:
|
||||
virtual ~SyscallHandler() = default;
|
||||
@@ -53,6 +77,9 @@ namespace FEXCore::HLE {
|
||||
|
||||
SyscallOSABI GetOSABI() const { return OSABI; }
|
||||
virtual FEXCore::CodeLoader *GetCodeLoader() const { return nullptr; }
|
||||
virtual void MarkGuestExecutableRange(uint64_t Start, uint64_t Length) { }
|
||||
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(uint64_t GuestAddr) = 0;
|
||||
virtual std::shared_lock<std::shared_mutex> CompileCodeLock(uint64_t Start) = 0;
|
||||
|
||||
protected:
|
||||
SyscallOSABI OSABI;
|
||||
|
||||
+2
-1
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXHeaderUtils/EnumOperators.h>
|
||||
|
||||
#include <array>
|
||||
@@ -560,7 +561,7 @@ class IRListView;
|
||||
class IREmitter;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<IREmitter> Parse(std::istream *in);
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, std::istream *in);
|
||||
|
||||
template<typename Type>
|
||||
inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
|
||||
+12
-3
@@ -19,11 +19,20 @@ friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
public:
|
||||
IREmitter()
|
||||
: DualListData {8 * 1024 * 1024} {
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024} {
|
||||
ReownOrClaimBuffer();
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
DualListData.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
IRListView ViewIR() { return IRListView(&DualListData, false); }
|
||||
IRListView *CreateIRCopy() { return new IRListView(&DualListData, true); }
|
||||
void ResetWorkingList();
|
||||
@@ -343,7 +352,7 @@ friend class FEXCore::IR::PassManager;
|
||||
OrderedNode *CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocator DualListData;
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
|
||||
OrderedNode *InvalidNode;
|
||||
OrderedNode *CurrentCodeBlock{};
|
||||
|
||||
+45
-15
@@ -3,6 +3,7 @@
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
@@ -20,21 +21,8 @@ namespace FEXCore::IR {
|
||||
*
|
||||
* Can potentially support reallocation if we are smart and make sure to invalidate anything holding a true pointer
|
||||
*/
|
||||
class DualIntrusiveAllocator final {
|
||||
class DualIntrusiveAllocator {
|
||||
public:
|
||||
DualIntrusiveAllocator() = delete;
|
||||
DualIntrusiveAllocator(DualIntrusiveAllocator &&) = delete;
|
||||
DualIntrusiveAllocator(size_t Size)
|
||||
: MemorySize {Size} {
|
||||
Data = reinterpret_cast<uintptr_t>(FEXCore::Allocator::malloc(Size * 2));
|
||||
List = reinterpret_cast<uintptr_t>(Data + Size);
|
||||
}
|
||||
|
||||
|
||||
~DualIntrusiveAllocator() {
|
||||
FEXCore::Allocator::free(reinterpret_cast<void*>(Data));
|
||||
}
|
||||
|
||||
[[nodiscard]] bool DataCheckSize(size_t Size) const {
|
||||
size_t NewOffset = DataCurrentOffset + Size;
|
||||
return NewOffset <= MemorySize;
|
||||
@@ -81,7 +69,11 @@ class DualIntrusiveAllocator final {
|
||||
memcpy(reinterpret_cast<void*>(List), reinterpret_cast<void*>(rhs.List), ListCurrentOffset);
|
||||
}
|
||||
|
||||
private:
|
||||
protected:
|
||||
DualIntrusiveAllocator(size_t Size)
|
||||
: MemorySize {Size} {
|
||||
}
|
||||
|
||||
uintptr_t Data;
|
||||
uintptr_t List;
|
||||
size_t DataCurrentOffset {0};
|
||||
@@ -89,6 +81,44 @@ class DualIntrusiveAllocator final {
|
||||
size_t MemorySize;
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorMalloc final : public DualIntrusiveAllocator {
|
||||
public:
|
||||
DualIntrusiveAllocatorMalloc(size_t Size)
|
||||
: DualIntrusiveAllocator {Size} {
|
||||
Data = reinterpret_cast<uintptr_t>(FEXCore::Allocator::malloc(Size * 2));
|
||||
List = reinterpret_cast<uintptr_t>(Data + Size);
|
||||
}
|
||||
|
||||
~DualIntrusiveAllocatorMalloc() {
|
||||
FEXCore::Allocator::free(reinterpret_cast<void*>(Data));
|
||||
}
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorThreadPool final : public DualIntrusiveAllocator {
|
||||
public:
|
||||
DualIntrusiveAllocatorThreadPool(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, size_t Size)
|
||||
: DualIntrusiveAllocator {Size}
|
||||
, PoolObject{ThreadAllocator, Size * 2} {
|
||||
// Claim a buffer on allocation
|
||||
PoolObject.ReownOrClaimBuffer();
|
||||
}
|
||||
|
||||
~DualIntrusiveAllocatorThreadPool() {
|
||||
PoolObject.UnclaimBuffer();
|
||||
}
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
Data = PoolObject.ReownOrClaimBuffer();
|
||||
List = Data + MemorySize;
|
||||
}
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
private:
|
||||
Utils::FixedSizePooledAllocation<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
class IRListView final {
|
||||
enum Flags {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
#include "IR.h"
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
@@ -46,6 +47,14 @@ class FEX_PACKED RegisterAllocationData {
|
||||
return sizeof(RegisterAllocationData) + NodeCount * sizeof(Map[0]);
|
||||
}
|
||||
|
||||
RegisterAllocationData* CreateCopy() {
|
||||
auto copy = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(MapCount));
|
||||
memcpy((void*)©->Map[0], (void*)&Map[0], MapCount * sizeof(Map[0]));
|
||||
copy->SpillSlotCount = SpillSlotCount;
|
||||
copy->MapCount = MapCount;
|
||||
copy->IsShared = IsShared;
|
||||
return copy;
|
||||
}
|
||||
void Serialize(std::ostream& stream) const {
|
||||
stream.write((const char*)&SpillSlotCount, sizeof(SpillSlotCount));
|
||||
stream.write((const char*)&MapCount, sizeof(MapCount));
|
||||
|
||||
@@ -0,0 +1,502 @@
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <cstddef>
|
||||
#include <list>
|
||||
#include <mutex>
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
/**
|
||||
* @brief An intrusive thread pool allocator
|
||||
*
|
||||
* Requires coordination between the allocator and its clients to efficiently share memory allocations between threads.
|
||||
*
|
||||
* The `Client` in this case referring to the location in code allocating a `MemoryBuffer` from the allocator.
|
||||
* - The client must `Claim` a buffer to allocate it
|
||||
* - In claiming a buffer, the allocator is passed a `BufferOwnedFlag` that is updated by both the allocator and client.
|
||||
* - When the client is done with the buffer it must `Disown` or `Unclaim` the buffer.
|
||||
* - `Disown` the buffer when it is expected to be used again soon.
|
||||
* - This is relatively cheap.
|
||||
* - `Unclaim` when the buffer won't be used again for an extended period.
|
||||
* - This is expensive and requires a mutex shared between threads
|
||||
* - `FixedSizePooledAllocation` helper class provided to help with this.
|
||||
*
|
||||
* Once the client has disowned a buffer then the allocator is free to reclaim the buffer when another thread is trying to `Claim` a new buffer.
|
||||
* The buffer getting claimed from a disowned client must have had its last use greater than the defined `DURATION` before it has a chance to get
|
||||
* reclaimed by the Allocator.
|
||||
*
|
||||
* During buffer reclaiming is also when unclaimed buffers get freed. This means active threads are able to clean up idle thread's unused memory.
|
||||
*/
|
||||
class IntrusivePooledAllocator {
|
||||
public:
|
||||
template<typename T>
|
||||
struct AllocationInfo {
|
||||
T Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
struct MemoryBuffer;
|
||||
/**
|
||||
* @brief Container for tracking the buffers
|
||||
*
|
||||
* We're using std::list explicitly because its iterators aren't invalidated when the list is adjusted.
|
||||
* if we had list types that we can atomically erase and append elements then unclaiming could be made cheaper.
|
||||
*/
|
||||
using ContainerType = std::list<MemoryBuffer*>;
|
||||
/**
|
||||
* @brief steady_clock to ensure long running applications don't hit any timeskip problems.
|
||||
*/
|
||||
using ClockType = std::chrono::steady_clock;
|
||||
/**
|
||||
* @brief Atomic flag state for letting the client know if it owns the buffer
|
||||
*/
|
||||
enum class ClientFlags : uint32_t {
|
||||
FLAG_FREE = 0,
|
||||
FLAG_OWNED = 1,
|
||||
FLAG_DISOWNED = 3,
|
||||
};
|
||||
|
||||
using BufferOwnedFlag = std::atomic<ClientFlags>;
|
||||
|
||||
struct MemoryBuffer {
|
||||
void* Ptr;
|
||||
size_t Size;
|
||||
std::atomic<std::chrono::time_point<ClockType>> LastUsed;
|
||||
BufferOwnedFlag *CurrentClientOwnedFlag{};
|
||||
};
|
||||
// Ensure that the atomic objects of MemoryBuffer are lock free
|
||||
static_assert(decltype(MemoryBuffer::LastUsed){}.is_always_lock_free, "Oops, needs to be lock free");
|
||||
static_assert(std::remove_pointer<decltype(MemoryBuffer::CurrentClientOwnedFlag)>::type{}.is_always_lock_free, "Oops, needs to be lock free");
|
||||
|
||||
/**
|
||||
* @brief Lets the client easily check if they own the buffer or not
|
||||
*
|
||||
* @param CurrentClientFlag Client owned flag
|
||||
*
|
||||
* @return Is the client buffer owned at the point of checking
|
||||
*/
|
||||
static bool IsClientBufferOwned(BufferOwnedFlag &CurrentClientFlag) {
|
||||
return CurrentClientFlag.load() == ClientFlags::FLAG_OWNED;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Lets the client easily check if the buffer was freed
|
||||
*
|
||||
* @param CurrentClientFlag Client owned flag
|
||||
*
|
||||
* @return Is the client buffer owned at the point of checking
|
||||
*/
|
||||
static bool IsClientBufferFree(BufferOwnedFlag &CurrentClientFlag) {
|
||||
return CurrentClientFlag.load() == ClientFlags::FLAG_FREE;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Allocates and claims a buffer that is tracked from the thread pool
|
||||
*
|
||||
* @param Size
|
||||
* @param CurrentClientFlag
|
||||
*
|
||||
* Once a buffer is claimed, the pool allocator can not reclaim this buffer until it is "Disowned"
|
||||
*
|
||||
* @return iterator to the internal tracking container
|
||||
*/
|
||||
ContainerType::iterator ClaimBuffer(size_t Size, BufferOwnedFlag *CurrentClientFlag) {
|
||||
std::unique_lock lk {AllocationMutex};
|
||||
auto Buffer = ClaimBufferImpl(Size);
|
||||
(*Buffer)->CurrentClientOwnedFlag = CurrentClientFlag;
|
||||
CurrentClientFlag->store(ClientFlags::FLAG_OWNED);
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Immediately release the buffer back to the allocator
|
||||
*
|
||||
* @param Buffer - The iterator that was previously given with ClaimBuffer
|
||||
*
|
||||
* Once this is called on a buffer then the pool allocator has full ownership of the buffer
|
||||
*/
|
||||
void UnclaimBuffer(ContainerType::iterator Buffer) {
|
||||
std::unique_lock lk {AllocationMutex};
|
||||
(*Buffer)->CurrentClientOwnedFlag->store(ClientFlags::FLAG_FREE);
|
||||
UnclaimBufferImpl(Buffer);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Set internal flags of buffer claiming that the buffer is relinquished ownership
|
||||
*
|
||||
* @param Buffer - The iterator that was previously given with ClaimBuffer
|
||||
*
|
||||
* Once the buffer is disowned, the allocator can take back ownership of the buffer at any time
|
||||
*
|
||||
* Use ReownOrClaimBuffer if you want to attempt reusing a buffer being held on to.
|
||||
*/
|
||||
void DisownBuffer(ContainerType::iterator Buffer) {
|
||||
// Client still owns the buffer but isn't using it
|
||||
// Allows us to claim it back if necessary
|
||||
(*Buffer)->LastUsed.store(ClockType::now(), std::memory_order_relaxed);
|
||||
(*Buffer)->CurrentClientOwnedFlag->store(ClientFlags::FLAG_DISOWNED);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Try to reown a buffer that we have previous disowned, failing that, claim a new buffer
|
||||
*
|
||||
* @param Buffer - The buffer we previously disowned
|
||||
* @param Size - The size of the buffer
|
||||
* @param CurrentClientFlag - The client tracked flag
|
||||
*
|
||||
* Once a DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always Reown a buffer after disowning it before use!
|
||||
*
|
||||
* @return Either the original buffer passed in if we managed to reclaim, or a new buffer if we couldn't
|
||||
*/
|
||||
ContainerType::iterator ReownOrClaimBuffer(ContainerType::iterator Buffer, size_t Size, BufferOwnedFlag *CurrentClientFlag) {
|
||||
ClientFlags Expected = ClientFlags::FLAG_DISOWNED;
|
||||
if (CurrentClientFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_OWNED)) {
|
||||
// If we managed to change the flag from DISOWNED to OWNED then we have successfully reclaimed
|
||||
// Finish setting up state
|
||||
(*Buffer)->LastUsed.store(ClockType::now(), std::memory_order_relaxed);
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
// Couldn't reclaim, just get a new buffer
|
||||
return ClaimBuffer(Size, CurrentClientFlag);
|
||||
}
|
||||
|
||||
virtual ~IntrusivePooledAllocator() = default;
|
||||
|
||||
// XXX: Is this a good amount?
|
||||
/**
|
||||
* @brief Duration before the allocator will reclaim buffers that the client claimed AND disowned
|
||||
*
|
||||
* Pool allocator will not attempt to reclaim client owned buffers, would be unsafe to do so.
|
||||
*/
|
||||
constexpr static std::chrono::duration DURATION {std::chrono::seconds(5)};
|
||||
|
||||
protected:
|
||||
IntrusivePooledAllocator() = default;
|
||||
|
||||
ContainerType::iterator ClaimBufferImpl(size_t Size) {
|
||||
auto BuffersEnd = UnclaimedBuffers.end();
|
||||
ContainerType::iterator BestFit = BuffersEnd;
|
||||
ContainerType::iterator UnsizedFit = BuffersEnd;
|
||||
|
||||
auto Now = ClockType::now();
|
||||
// Move any expired ClaimedBuffers to UnclaimedBuffers
|
||||
{
|
||||
// Spin the non-owned buffers and see if we can take ones past the period
|
||||
for (auto it = ClaimedBuffers.begin(); it != ClaimedBuffers.end();) {
|
||||
// 1) Can't take anything that the client has still claimed
|
||||
// 2) Needs to still be last used beyond our time threshold
|
||||
// 3) Only take the oldest buffer
|
||||
if ((*it)->CurrentClientOwnedFlag->load() == ClientFlags::FLAG_DISOWNED) {
|
||||
auto UsedTime = (*it)->LastUsed.load(std::memory_order_relaxed);
|
||||
if ((Now - UsedTime) >= DURATION) {
|
||||
ClientFlags Expected = ClientFlags::FLAG_DISOWNED;
|
||||
if ((*it)->CurrentClientOwnedFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_FREE)) {
|
||||
// We managed to take away ownership
|
||||
// Put it back in the regular pool and come back to it
|
||||
(*it)->CurrentClientOwnedFlag = nullptr;
|
||||
UnclaimedBuffers.emplace_back(*it);
|
||||
it = ClaimedBuffers.erase(it);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
++it;
|
||||
}
|
||||
}
|
||||
|
||||
// Find an unclaimed buffer that is >= Size and Free up to one unclaimed buffer that has expired
|
||||
{
|
||||
// Walk all the allocations and find a buffer that fits
|
||||
for (auto it = UnclaimedBuffers.begin(); it != BuffersEnd; ++it) {
|
||||
if ((*it)->Size == Size) {
|
||||
BestFit = it;
|
||||
break;
|
||||
}
|
||||
|
||||
if ((*it)->Size > Size) {
|
||||
UnsizedFit = it;
|
||||
}
|
||||
}
|
||||
|
||||
// If we didn't have an exact fit then use an unsized fit
|
||||
if (BestFit == BuffersEnd) {
|
||||
BestFit = UnsizedFit;
|
||||
}
|
||||
|
||||
// Free up to one unclaimed buffer that has expired
|
||||
{
|
||||
std::chrono::time_point<ClockType> LRUTime{};
|
||||
ContainerType::iterator LastUsed = BuffersEnd;
|
||||
|
||||
// Walk all the allocations and find a buffer to erase
|
||||
for (auto it = UnclaimedBuffers.begin(); it != UnclaimedBuffers.end(); ++it) {
|
||||
// Ensure that the LRU value is past our duration threshold and isn't the one we are claiming
|
||||
// Also only select a single memory region
|
||||
if (it != BestFit) {
|
||||
auto UsedTime = (*it)->LastUsed.load(std::memory_order_relaxed);
|
||||
if ((Now - UsedTime) >= DURATION &&
|
||||
UsedTime > LRUTime) {
|
||||
LastUsed = it;
|
||||
LRUTime = UsedTime;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// If we found a buffer then free it
|
||||
if (LastUsed != BuffersEnd) {
|
||||
Free((*LastUsed)->Ptr, (*LastUsed)->Size);
|
||||
delete *LastUsed;
|
||||
UnclaimedBuffers.erase(LastUsed);
|
||||
}
|
||||
}
|
||||
|
||||
if (BestFit != UnclaimedBuffers.end()) {
|
||||
MemoryBuffer *Buffer = *BestFit;
|
||||
UnclaimedBuffers.erase(BestFit);
|
||||
return ClaimedBuffers.emplace(ClaimedBuffers.end(), Buffer);
|
||||
}
|
||||
}
|
||||
|
||||
// Need to allocate a new buffer, couldn't fit
|
||||
auto Data = Alloc(Size);
|
||||
return ClaimedBuffers.emplace(ClaimedBuffers.end(), new MemoryBuffer{Data, Size, ClockType::now()});
|
||||
}
|
||||
|
||||
void UnclaimBufferImpl(ContainerType::iterator Buffer) {
|
||||
(*Buffer)->CurrentClientOwnedFlag = nullptr;
|
||||
UnclaimedBuffers.emplace_back(*Buffer);
|
||||
ClaimedBuffers.erase(Buffer);
|
||||
}
|
||||
|
||||
void FreeAllBuffers() {
|
||||
for (auto it : UnclaimedBuffers) {
|
||||
Free(it->Ptr, it->Size);
|
||||
delete it;
|
||||
}
|
||||
|
||||
for (auto it : ClaimedBuffers) {
|
||||
Free(it->Ptr, it->Size);
|
||||
delete it;
|
||||
}
|
||||
|
||||
UnclaimedBuffers.clear();
|
||||
ClaimedBuffers.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief List of buffers that this pool allocator itself owns
|
||||
*/
|
||||
ContainerType UnclaimedBuffers;
|
||||
|
||||
/**
|
||||
* @brief List of buffers that are client claimed
|
||||
*/
|
||||
ContainerType ClaimedBuffers;
|
||||
|
||||
/**
|
||||
* @brief Mutex to ensure thread safety while shuffling buffers around and allocating
|
||||
*/
|
||||
std::mutex AllocationMutex;
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Allocates the buffer
|
||||
*
|
||||
* @param Size of the object to allocate
|
||||
*
|
||||
* @return pointer
|
||||
*/
|
||||
virtual void *Alloc(size_t Size) = 0;
|
||||
/**
|
||||
* @brief Frees the buffer
|
||||
*
|
||||
* @param Ptr buffer pointer
|
||||
* @param Size buffer size
|
||||
*/
|
||||
virtual void Free(void* Ptr, size_t Size) = 0;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Thread pool allocator that allocates and frees objects using malloc
|
||||
*/
|
||||
class PooledAllocatorMalloc final : public IntrusivePooledAllocator {
|
||||
public:
|
||||
PooledAllocatorMalloc() = default;
|
||||
|
||||
virtual ~PooledAllocatorMalloc() {
|
||||
FreeAllBuffers();
|
||||
}
|
||||
|
||||
private:
|
||||
void *Alloc(size_t Size) override {
|
||||
return FEXCore::Allocator::malloc(Size);
|
||||
}
|
||||
|
||||
void Free(void* Ptr, size_t Size) override {
|
||||
FEXCore::Allocator::free(Ptr);
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Thread pool allocator that allocates and frees objects that uses mmap
|
||||
*/
|
||||
class PooledAllocatorMMap final : public IntrusivePooledAllocator {
|
||||
public:
|
||||
PooledAllocatorMMap() = default;
|
||||
|
||||
virtual ~PooledAllocatorMMap() {
|
||||
FreeAllBuffers();
|
||||
}
|
||||
|
||||
private:
|
||||
void *Alloc(size_t Size) override {
|
||||
return FEXCore::Allocator::mmap(0, Size,
|
||||
PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
}
|
||||
|
||||
void Free(void* Ptr, size_t Size) override {
|
||||
FEXCore::Allocator::munmap(Ptr, Size);
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Wrapper around the pool allocator for delayed pool reclaiming
|
||||
*
|
||||
* This is expected to be used in high frequency buffer temporary usage.
|
||||
* Instead of quickly unclaiming and reclaiming the buffer while the the code is hot,
|
||||
* This instead will do the cheap operation of disowning the buffer until the code path cools down enough.
|
||||
* Once the code path stops disowning the codepath more times than `PeriodFrequency` during `PeriodMS` then
|
||||
* it will immediately unclaim.
|
||||
*
|
||||
* Implications:
|
||||
* - The object will always be claimed for at *least* `PeriodFrequency`
|
||||
* - The object will still *always* be disowned after each temporary use
|
||||
* - This allows the pool allocator to reclaim a buffer from a sleeping thread
|
||||
*
|
||||
* Performance characteristics:
|
||||
* - Disowning is cheap.
|
||||
* - Last-used timestamp update
|
||||
* - atomic_bool clear to signify it is disowned
|
||||
*
|
||||
* - Reowning is relatively cheap (When buffer is still owned).
|
||||
* - atomic_bool load to check if the object is still owned
|
||||
* - atomic<uint32_t> CAS to change the object to `OWNED` state
|
||||
* - Resolves a race condition where the `Allocator` can be in the process of reclaiming the buffer from the client
|
||||
* - Last-used timestamp update
|
||||
* - atomic_bool<relaxed> set to signify owned
|
||||
* - atomic<uint32_t> set to change object to `OWNED` state
|
||||
* - When object isn't owned, then allocate a new buffer from the pool
|
||||
*
|
||||
* - Unclaiming is fairly costly
|
||||
* - Requires owning a mutex, shared between all threads using the `Allocator`
|
||||
* - Updating two std::list containers to give the ownership back to the `Allocator`
|
||||
*
|
||||
* - Claiming is very costly
|
||||
* - Requires owning a mutex, shared between all threads using the `Allocator`
|
||||
* - Scans two std::list containers to find the best fit buffer
|
||||
* - Or allocates another buffer when that fails
|
||||
* - Frees stale buffers opportunistically
|
||||
*/
|
||||
template<typename Type, size_t PeriodMS, size_t PeriodFrequency>
|
||||
class FixedSizePooledAllocation final {
|
||||
// If the delayed object reclaimer is more than the thread pool allocator's duration then the pool allocator would always need to reclaim the
|
||||
// buffer rather than giving it back.
|
||||
static_assert(std::chrono::duration(std::chrono::milliseconds(PeriodMS)) <= IntrusivePooledAllocator::DURATION,
|
||||
"DeplayedObjectReclaimer period needs to be lower or equal to the pool allocator duration");
|
||||
|
||||
public:
|
||||
FixedSizePooledAllocation(IntrusivePooledAllocator &Allocator, size_t Size)
|
||||
: ThreadAllocator {Allocator}
|
||||
, Size {Size} {
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Return the owned buffer or allocate another one from the `Allocator`
|
||||
*
|
||||
* The buffer returned isn't guaranteed to be the exact size of `Size` but it will be at least `Size`.
|
||||
* The contents of the memory returned isn't guaranteed to be zero initialized or not.
|
||||
* Not even guaranteed to contain the previous data from the previous reowning if the pointer is the same.
|
||||
*
|
||||
* @return object of type `Type` allocated with at least the size of `Size` from the constructor
|
||||
*/
|
||||
Type ReownOrClaimBuffer() {
|
||||
if (!FEXCore::Utils::IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag)) {
|
||||
Info = ThreadAllocator.ReownOrClaimBuffer(Info, Size, &ClientOwnedFlag);
|
||||
}
|
||||
|
||||
// Putting a memset here is very handy for using thread sanitizer to find buffer usage races
|
||||
// Leaving this here for future excavation that will definitely occur here
|
||||
// memset((*Info)->Ptr, 0, Size);
|
||||
|
||||
return reinterpret_cast<Type>((*Info)->Ptr);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Disown or unclaim the buffer, letting the `Allocator` know it can reclaim the buffer
|
||||
*
|
||||
* Once the `ReownOrClaimBuffer` function has been used, this must be called to let the `Allocator` know it is safe to reclaim a buffer.
|
||||
*
|
||||
* This will first Disown the buffer; which is cheap.
|
||||
*
|
||||
* If the frequency of use is below the threshold then immediately `UnclaimBuffer` so that `Allocator` can reuse it.
|
||||
*/
|
||||
void DelayedDisownBuffer() {
|
||||
LOGMAN_THROW_A_FMT(FEXCore::Utils::IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag),
|
||||
"Tried to disown buffer when client doesn't own it");
|
||||
|
||||
// Always disown but not always unclaim
|
||||
// Disowning = cheap, unclaiming = expensive
|
||||
ThreadAllocator.DisownBuffer(Info);
|
||||
|
||||
auto Now = std::chrono::steady_clock::now();
|
||||
if ((Now - Previous) >= std::chrono::duration(std::chrono::milliseconds(PeriodMS))) {
|
||||
if (CountPer < PeriodFrequency) {
|
||||
// Only unclaim the buffer if our buffer usage isn't excessive in the last period
|
||||
UnclaimBuffer();
|
||||
}
|
||||
CountPer = 0;
|
||||
Previous = Now;
|
||||
}
|
||||
++CountPer;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Completely unclaim the buffer
|
||||
*
|
||||
* Useful if it is known that the buffer won't be used again for a period and can be given back
|
||||
* to the `Allocator` immediately.
|
||||
*
|
||||
* Necessary if an object is going to be freed from memory, so the `Allocator` can't update the `ClientOwnedFlag`
|
||||
*
|
||||
* Only use in that edge case! Otherwise use `DelayedDisownBuffer`
|
||||
*/
|
||||
void UnclaimBuffer() {
|
||||
if (!FEXCore::Utils::IntrusivePooledAllocator::IsClientBufferFree(ClientOwnedFlag)) {
|
||||
ThreadAllocator.UnclaimBuffer(Info);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Thread allocator
|
||||
FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator;
|
||||
|
||||
// Buffer size
|
||||
size_t Size;
|
||||
|
||||
// Buffer ownership tracking
|
||||
FEXCore::Utils::IntrusivePooledAllocator::ContainerType::iterator Info{};
|
||||
FEXCore::Utils::IntrusivePooledAllocator::BufferOwnedFlag ClientOwnedFlag { FEXCore::Utils::IntrusivePooledAllocator::ClientFlags::FLAG_FREE };
|
||||
|
||||
// Threshold counting
|
||||
uint64_t CountPer{};
|
||||
std::chrono::steady_clock::time_point Previous;
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,265 @@
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <linux/futex.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <shared_mutex>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
/**
|
||||
* @brief This class is similar to std::shared_mutex but is safe to shared lock multiple times from the same thread.
|
||||
*
|
||||
* Just like std::shared_mutex, this has shared lock priority when a shared lock is already held.
|
||||
*/
|
||||
class refcount_shared_mutex final {
|
||||
public:
|
||||
void lock() {
|
||||
auto UniqueResult = TryUniqueLock();
|
||||
if (UniqueResult.second) {
|
||||
// Managed to get the unique lock
|
||||
return;
|
||||
}
|
||||
|
||||
int Op = FUTEX_WAIT | FUTEX_PRIVATE_FLAG;
|
||||
|
||||
do {
|
||||
::syscall(SYS_futex,
|
||||
&Futex,
|
||||
Op,
|
||||
UniqueResult.first, // Value
|
||||
nullptr, // Timeout
|
||||
nullptr, // Addr
|
||||
0);
|
||||
|
||||
UniqueResult = TryUniqueLock();
|
||||
// If Res == 0 then check the unique lock to see if unique is no longer owned
|
||||
if (UniqueResult.second) {
|
||||
// Unique lock succeeded
|
||||
return;
|
||||
}
|
||||
} while (true);
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
auto UniqueResult = TryUniqueLock();
|
||||
return UniqueResult.second;
|
||||
}
|
||||
|
||||
void unlock() {
|
||||
LOGMAN_THROW_A_FMT(Futex.load() == UNIQUE_LOCK_VALUE, "Tried unlocking not locked mutex?");
|
||||
|
||||
auto TryUniqueUnlock = [this]() -> std::pair<uint32_t, bool> {
|
||||
auto LocalFutex = Futex.load();
|
||||
|
||||
if (LocalFutex != UNIQUE_LOCK_VALUE) {
|
||||
// Refcount must be zero if we are to attempt getting a unique lock
|
||||
}
|
||||
else {
|
||||
// Try locking now in userspace
|
||||
while (LocalFutex == UNIQUE_LOCK_VALUE) {
|
||||
auto Desired = LocalFutex;
|
||||
Desired = 0;
|
||||
if (Futex.compare_exchange_strong(LocalFutex, Desired)) {
|
||||
// We have successfully unique locked
|
||||
return std::make_pair(Desired, true);
|
||||
}
|
||||
else {
|
||||
if (LocalFutex == UNIQUE_LOCK_VALUE) {
|
||||
// If another thread pulled the unique lock or the ref count incremented
|
||||
// Then we need to wait, loop will end
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return std::make_pair(LocalFutex, false);
|
||||
};
|
||||
|
||||
[[maybe_unused]] auto UniqueResult = TryUniqueUnlock();
|
||||
LOGMAN_THROW_A_FMT(UniqueResult.second, "Couldn't unlock mutex memory?");
|
||||
|
||||
// We've now unlocked, use the futex to wake up any shared waiters
|
||||
int Op = FUTEX_WAKE | FUTEX_PRIVATE_FLAG;
|
||||
::syscall(SYS_futex,
|
||||
&Futex,
|
||||
Op,
|
||||
INT_MAX, // Could be any number of shared waiters
|
||||
nullptr, // timeout
|
||||
nullptr, // addr
|
||||
0);
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
return TryRefIncrement();
|
||||
}
|
||||
|
||||
void lock_shared() {
|
||||
if (TryRefIncrement()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Unique lock was held. Wait until it is no longer held using a system futex
|
||||
int Op = FUTEX_WAIT | FUTEX_PRIVATE_FLAG;
|
||||
|
||||
auto Expected = UNIQUE_LOCK_VALUE;
|
||||
do {
|
||||
::syscall(SYS_futex,
|
||||
&Futex,
|
||||
Op,
|
||||
Expected, // Value
|
||||
nullptr, // Timeout,
|
||||
nullptr, // Addr
|
||||
0);
|
||||
|
||||
Expected = Futex.load();
|
||||
// If Res == 0 then check the unique lock to see if unique is no longer owned
|
||||
if (Expected != UNIQUE_LOCK_VALUE) {
|
||||
if (TryRefIncrement()) {
|
||||
// Ref count succeeded
|
||||
return;
|
||||
}
|
||||
}
|
||||
} while (true);
|
||||
}
|
||||
|
||||
// Returns the number of ref counts remaining once this leaves
|
||||
uint32_t unlock_shared() {
|
||||
auto DecrementResult = TryRefDecrement();
|
||||
|
||||
if (DecrementResult.second) {
|
||||
if (DecrementResult.first == 0) {
|
||||
// If we were the last shared value out then we need to do a futex to wake up any waiters
|
||||
int Op = FUTEX_WAKE | FUTEX_PRIVATE_FLAG;
|
||||
::syscall(SYS_futex,
|
||||
&Futex,
|
||||
Op,
|
||||
1, // Wake up only one thread if one is waiting. Which would be the unique waiter
|
||||
nullptr, // timeout
|
||||
nullptr, // addr
|
||||
0);
|
||||
}
|
||||
return DecrementResult.first;
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Managed to squeeze a unique lock between shared locks?");
|
||||
return 0; // Error
|
||||
}
|
||||
|
||||
// Get the raw futex ref count number
|
||||
uint32_t GetNumRefCounts() const {
|
||||
return Futex.load();
|
||||
}
|
||||
|
||||
// Be careful with this. Only use when you know the mutex is dead
|
||||
void Reset() {
|
||||
Futex.store(0);
|
||||
|
||||
int Op = FUTEX_WAKE | FUTEX_PRIVATE_FLAG;
|
||||
::syscall(SYS_futex,
|
||||
&Futex,
|
||||
Op,
|
||||
INT_MAX, // Wake up all threads if any waiting
|
||||
nullptr, // timeout
|
||||
nullptr, // addr
|
||||
0);
|
||||
}
|
||||
|
||||
private:
|
||||
bool TryRefIncrement() {
|
||||
auto LocalFutex = Futex.load();
|
||||
|
||||
if (LocalFutex == UNIQUE_LOCK_VALUE) {
|
||||
// Unique lock held
|
||||
}
|
||||
else {
|
||||
// Try to increment the counter if unique lock isn't held
|
||||
while (LocalFutex != UNIQUE_LOCK_VALUE) {
|
||||
auto Desired = LocalFutex;
|
||||
Desired++;
|
||||
|
||||
// Try to increment the ref count
|
||||
if (Futex.compare_exchange_strong(LocalFutex, Desired)) {
|
||||
// We have successfully incremented the ref counting mutex in userspace
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
if (LocalFutex == UNIQUE_LOCK_VALUE) {
|
||||
// Unique lock was held
|
||||
// Nothing to do, loop will end
|
||||
}
|
||||
|
||||
// Try again. Can happen in a race to increment the ref count
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
};
|
||||
|
||||
std::pair<uint32_t, bool> TryRefDecrement() {
|
||||
auto LocalFutex = Futex.load();
|
||||
|
||||
if (LocalFutex == UNIQUE_LOCK_VALUE) {
|
||||
// Unique lock held
|
||||
}
|
||||
else {
|
||||
// Try to increment the counter if unique lock isn't held
|
||||
while (LocalFutex != UNIQUE_LOCK_VALUE) {
|
||||
auto Desired = LocalFutex;
|
||||
Desired--;
|
||||
|
||||
// Try to increment the ref count
|
||||
if (Futex.compare_exchange_strong(LocalFutex, Desired)) {
|
||||
// We have successfully incremented the ref counting mutex in userspace
|
||||
return std::make_pair(Desired, true);
|
||||
}
|
||||
else {
|
||||
if (LocalFutex == UNIQUE_LOCK_VALUE) {
|
||||
// Unique lock was held
|
||||
// Nothing to do, loop will end
|
||||
}
|
||||
|
||||
// Try again. Can happen in a race to increment the ref count
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return std::make_pair(LocalFutex, false);
|
||||
};
|
||||
|
||||
std::pair<uint32_t, bool> TryUniqueLock() {
|
||||
auto LocalFutex = Futex.load();
|
||||
|
||||
if (LocalFutex) {
|
||||
// Refcount must be zero if we are to attempt getting a unique lock
|
||||
}
|
||||
else {
|
||||
// Try locking now in userspace
|
||||
while (LocalFutex == 0) {
|
||||
auto Desired = LocalFutex;
|
||||
Desired = UNIQUE_LOCK_VALUE;
|
||||
if (Futex.compare_exchange_strong(LocalFutex, Desired)) {
|
||||
// We have successfully unique locked
|
||||
return std::make_pair(Desired, true);
|
||||
}
|
||||
else {
|
||||
if (LocalFutex == 0) {
|
||||
// If another thread pulled the unique lock or the ref count incremented
|
||||
// Then we need to wait, loop will end
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return std::make_pair(LocalFutex, false);
|
||||
};
|
||||
|
||||
constexpr static uint32_t UNIQUE_LOCK_VALUE = -4096U;
|
||||
// -1 = unique_lock
|
||||
// 0 = no shared
|
||||
// >0 = shared waiters
|
||||
std::atomic<uint32_t> Futex{};
|
||||
};
|
||||
}
|
||||
Vendored
+1
-1
Submodule External/drm-headers updated: 4d7eed48e8...97e7ed6fe4.
@@ -5,13 +5,22 @@
|
||||
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FHU {
|
||||
/**
|
||||
* @brief A class that masks signals and locks a mutex until it goes out of scope
|
||||
* @brief A drop-in replacement for std::lock_guard that masks POSIX signals while the mutex is locked
|
||||
*
|
||||
* Use this class to prevent reentrancy issues of C++ mutexes with certain signal handlers.
|
||||
* Common examples of such issues are:
|
||||
* - C++ mutexes not unlocking due to a signal handler longjmping out of a scope owning the mutex
|
||||
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
|
||||
* again before unlocking
|
||||
*
|
||||
* Ownership of this object may be moved, but it is NOT SAFE to move across threads.
|
||||
*
|
||||
* Constructor order:
|
||||
* 1) Mask signals
|
||||
@@ -21,26 +30,44 @@ namespace FHU {
|
||||
* 1) Unlock Mutex
|
||||
* 2) Unmask signals
|
||||
*/
|
||||
class ScopedSignalMaskWithMutex final {
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedSignalMaskWithMutexBase final {
|
||||
public:
|
||||
ScopedSignalMaskWithMutex(std::mutex &_Mutex, uint64_t Mask = ~0ULL)
|
||||
: Mutex {_Mutex} {
|
||||
|
||||
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex} {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
|
||||
// Lock the mutex
|
||||
Mutex.lock();
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
~ScopedSignalMaskWithMutex() {
|
||||
// Unlock the mutex
|
||||
Mutex.unlock();
|
||||
// No copy or assignment possible
|
||||
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
|
||||
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
|
||||
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
// Only move
|
||||
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
|
||||
: OriginalMask {rhs.OriginalMask}, Mutex {rhs.Mutex} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedSignalMaskWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
}
|
||||
}
|
||||
private:
|
||||
uint64_t OriginalMask{};
|
||||
std::mutex &Mutex;
|
||||
MutexType *Mutex;
|
||||
};
|
||||
|
||||
using ScopedSignalMaskWithMutex = ScopedSignalMaskWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedSignalMaskWithSharedLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithUniqueLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
#pragma once
|
||||
|
||||
#ifndef DO_PRAGMA
|
||||
#define DO_PRAGMA(x) _Pragma (#x)
|
||||
#endif
|
||||
|
||||
#if FEX_WARN_TODO
|
||||
// FEX_TODO_ISSUE(github ticket number, "comment")
|
||||
#define FEX_TODO_ISSUE(github_ticket, comment) DO_PRAGMA(GCC warning "TODO: https://github.com/FEX-Emu/FEX/issues/" #github_ticket comment);
|
||||
// FEX_TODO("comment")
|
||||
#define FEX_TODO(comment) DO_PRAGMA(GCC warning "TODO: " comment);
|
||||
#else
|
||||
// FEX_TODO_ISSUE(github ticket number, "comment")
|
||||
#define FEX_TODO_ISSUE(github_ticket, comment)
|
||||
// FEX_TODO("comment")
|
||||
#define FEX_TODO(comment)
|
||||
#endif
|
||||
|
||||
// For linking to tickets, non-todo
|
||||
// FEX_TICKET(github ticket number) or FEX_TICKET(github ticket number, "comment")
|
||||
#define FEX_TICKET(github_ticket, ...)
|
||||
@@ -6,6 +6,16 @@ import logging
|
||||
logger = logging.getLogger()
|
||||
logger.setLevel(logging.WARNING)
|
||||
|
||||
# Usage of this script is `Scripts/GenerateSyscallNumbers.py <Path to Linux directory>`
|
||||
# This will then parse the syscall headers and format them in an enum
|
||||
# Then this will be output in stdout
|
||||
# This output should then be checked and copied to the following headers, splitting up the enums:
|
||||
# - Source/Tests/LinuxSyscalls/x32/SyscallsEnum.h
|
||||
# - Source/Tests/LinuxSyscalls/x64/SyscallsEnum.h
|
||||
# - Source/Tests/LinuxSyscalls/Arm64/SyscallsEnum.h
|
||||
# `FEX_Syscalls_Common` is provided in the output as just an indicator for which syscalls are using the common
|
||||
# syscall interface.
|
||||
|
||||
@dataclass
|
||||
class SyscallDefinition:
|
||||
arch: str
|
||||
@@ -44,6 +54,14 @@ Syscallx64File = "/arch/x86/entry/syscalls/syscall_64.tbl"
|
||||
Syscallx86File = "/arch/x86/entry/syscalls/syscall_32.tbl"
|
||||
SyscallArm64File = "/include/uapi/asm-generic/unistd.h"
|
||||
|
||||
# Syscall names that had naming conflict with some global definitions
|
||||
# Renamed to work around that issue
|
||||
DefinitionRenameDict = {
|
||||
"pread64": "pread_64",
|
||||
"pwrite64": "pwrite_64",
|
||||
"prlimit64": "prlimit_64"
|
||||
}
|
||||
|
||||
Definitions_x64 = []
|
||||
Definitions_x64_dict = {}
|
||||
Definitions_x86 = []
|
||||
@@ -86,6 +104,9 @@ def ParseArchSyscalls(Defs, DefsDict, Arch, FilePath, IgnoreArch):
|
||||
else:
|
||||
EntryName = split_text[3]
|
||||
|
||||
if Name in DefinitionRenameDict:
|
||||
Name = DefinitionRenameDict[Name]
|
||||
|
||||
Def = SyscallDefinition(Arch, Num, ABI, Name, EntryName)
|
||||
|
||||
Defs.append(Def)
|
||||
@@ -154,6 +175,9 @@ def ParseCommonArchSyscalls(Defs, DefsDict, Arch, FilePath):
|
||||
ABI = Arch
|
||||
EntryName = split_text[1].strip().split(")")[0]
|
||||
|
||||
if Name in DefinitionRenameDict:
|
||||
Name = DefinitionRenameDict[Name]
|
||||
|
||||
Def = SyscallDefinition(Arch, Num, ABI, Name, EntryName)
|
||||
|
||||
Defs.append(Def)
|
||||
|
||||
@@ -280,7 +280,7 @@ public:
|
||||
}
|
||||
|
||||
bool MapMemory(const MapperFn& Mapper, const UnmapperFn& Unmapper) override {
|
||||
auto DoMMap = [Mapper](uint64_t Address, size_t Size, bool FixedNoReplace) -> void* {
|
||||
auto DoMMap = [&Mapper](uint64_t Address, size_t Size, bool FixedNoReplace) -> void* {
|
||||
void *Result = Mapper(reinterpret_cast<void*>(Address), Size, PROT_READ | PROT_WRITE, (FixedNoReplace ? MAP_FIXED_NOREPLACE : MAP_FIXED) | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
LOGMAN_THROW_A_FMT(Result != (void*)~0ULL, "Couldn't mmap");
|
||||
return Result;
|
||||
|
||||
@@ -255,6 +255,11 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
Args.erase(Args.begin());
|
||||
}
|
||||
|
||||
// Append any additional arguments from config
|
||||
for (auto &Arg : AdditionalArguments.All()) {
|
||||
Args.emplace_back(Arg);
|
||||
}
|
||||
|
||||
if (!MainElf.InterpreterElf.empty()) {
|
||||
if (!InterpElf.ReadElf(ResolveRootfsFile(MainElf.InterpreterElf, RootFS)) && !InterpElf.ReadElf(MainElf.InterpreterElf))
|
||||
return;
|
||||
@@ -651,5 +656,6 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
uint64_t ArgumentBackingSize{};
|
||||
uint64_t EnvironmentBackingSize{};
|
||||
uint64_t BaseOffset{};
|
||||
FEX_CONFIG_OPT(AdditionalArguments, ADDITIONALARGUMENTS);
|
||||
|
||||
};
|
||||
@@ -59,7 +59,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
// Lets us start an emulated bash instance
|
||||
const size_t FEXArgsCount = std::size(FEXArgs) - (Args.empty() ? 1 : 0);
|
||||
|
||||
Argv.resize(Args.size() + FEXArgsCount + 1);
|
||||
Argv.resize(Args.size() + FEXArgsCount);
|
||||
|
||||
// Pass in the FEXInterpreter arguments
|
||||
for (size_t i = 0; i < FEXArgsCount; ++i) {
|
||||
@@ -70,7 +70,39 @@ int main(int argc, char **argv, char **const envp) {
|
||||
for (size_t i = 0; i < Args.size(); ++i) {
|
||||
Argv[i + FEXArgsCount] = Args[i].c_str();
|
||||
}
|
||||
Argv[Argv.size() - 1] = nullptr;
|
||||
|
||||
return execve(Argv[0], const_cast<char *const*>(&Argv.at(0)), envp);
|
||||
// Set --norc when no arguments are passed so PS1 doesn't get overwritten
|
||||
const char* NoRC = "--norc";
|
||||
if (Args.empty()) {
|
||||
Argv.emplace_back(NoRC);
|
||||
}
|
||||
|
||||
Argv.emplace_back(nullptr);
|
||||
|
||||
// Prepend `FEXBash>` to PS1 to be less confusing about running under emulation
|
||||
// In most cases PS1 isn't an environment variable, but instead a shell variable
|
||||
// But in case the user has set the PS1 environment variable then still prepend
|
||||
//
|
||||
// To get the shell variables as an environment variable then you can do `PS1=$PS1 FEXBash`
|
||||
std::vector<const char *> Envp{};
|
||||
char *PS1Env{};
|
||||
for (unsigned i = 0;; ++i) {
|
||||
if (envp[i] == nullptr)
|
||||
break;
|
||||
if (strstr(envp[i], "PS1=") == envp[i]) {
|
||||
PS1Env = envp[i];
|
||||
}
|
||||
else {
|
||||
Envp.emplace_back(envp[i]);
|
||||
}
|
||||
}
|
||||
|
||||
std::string PS1 = "PS1=FEXBash> ";
|
||||
if (PS1Env) {
|
||||
PS1 += &PS1Env[strlen("PS1=")];
|
||||
}
|
||||
Envp.emplace_back(PS1.c_str());
|
||||
Envp.emplace_back(nullptr);
|
||||
|
||||
return execve(Argv[0], const_cast<char *const*>(&Argv.at(0)), const_cast<char *const*>(&Envp[0]));
|
||||
}
|
||||
+11
-24
@@ -328,15 +328,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (Loader.Is64BitMode()) {
|
||||
// Destroy the 48th bit if it exists
|
||||
Base48Bit = FEXCore::Allocator::Steal48BitVA();
|
||||
if (!Loader.MapMemory([](void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
return FEXCore::Allocator::mmap(addr, length, prot, flags, fd, offset);
|
||||
}, [](void *addr, size_t length) {
|
||||
return FEXCore::Allocator::munmap(addr, length);
|
||||
})) {
|
||||
// failed to map
|
||||
LogMan::Msg::EFmt("Failed to map 64-bit elf file.");
|
||||
return -ENOEXEC;
|
||||
}
|
||||
} else {
|
||||
FEX_CONFIG_OPT(Use32BitAllocator, FORCE32BITALLOCATOR);
|
||||
if (KernelVersion < FEX::HLE::SyscallHandler::KernelVersion(4, 17)) {
|
||||
@@ -355,16 +346,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
else {
|
||||
Allocator = FEX::HLE::CreatePassthroughAllocator();
|
||||
}
|
||||
|
||||
if (!Loader.MapMemory([&Allocator](void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
return Allocator->mmap(addr, length, prot, flags, fd, offset);
|
||||
}, [&Allocator](void *addr, size_t length) {
|
||||
return Allocator->munmap(addr, length);
|
||||
})) {
|
||||
// failed to map
|
||||
LogMan::Msg::EFmt("Failed to map 32-bit elf file.");
|
||||
return -ENOEXEC;
|
||||
}
|
||||
}
|
||||
|
||||
// System allocator is now system allocator or FEX
|
||||
@@ -396,6 +377,15 @@ int main(int argc, char **argv, char **const envp) {
|
||||
auto SyscallHandler = Loader.Is64BitMode() ? FEX::HLE::x64::CreateHandler(CTX, SignalDelegation.get())
|
||||
: FEX::HLE::x32::CreateHandler(CTX, SignalDelegation.get(), std::move(Allocator));
|
||||
|
||||
auto Mapper = std::bind_front(&FEX::HLE::SyscallHandler::GuestMmap, SyscallHandler.get());
|
||||
auto Unmapper = std::bind_front(&FEX::HLE::SyscallHandler::GuestMunmap, SyscallHandler.get());
|
||||
|
||||
if (!Loader.MapMemory(Mapper, Unmapper)) {
|
||||
// failed to map
|
||||
LogMan::Msg::EFmt("Failed to map %d-bit elf file.", Loader.Is64BitMode() ? 64 : 32);
|
||||
return -ENOEXEC;
|
||||
}
|
||||
|
||||
SyscallHandler->SetCodeLoader(&Loader);
|
||||
|
||||
auto BRKInfo = Loader.GetBRKInfo();
|
||||
@@ -450,10 +440,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
std::filesystem::rename(TmpFilepath, NewFilepath);
|
||||
});
|
||||
|
||||
for(const auto &Section: Loader.Sections) {
|
||||
FEXCore::Context::AddNamedRegion(CTX, Section.Base, Section.Size, Section.Offs, Section.Filename);
|
||||
}
|
||||
|
||||
if (AOTIRGenerate()) {
|
||||
for(auto &Section: Loader.Sections) {
|
||||
FEX::AOT::AOTGenSection(CTX, Section);
|
||||
@@ -462,7 +448,8 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Context::RunUntilExit(CTX);
|
||||
}
|
||||
|
||||
if (std::filesystem::create_directories(std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir", ec)) {
|
||||
std::filesystem::create_directories(std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir", ec);
|
||||
if (!ec) {
|
||||
FEXCore::Context::WriteFilesWithCode(CTX, [](const std::string& fileid, const std::string& filename) {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".path");
|
||||
int fd = open(filepath.c_str(), O_CREAT | O_EXCL | O_WRONLY, 0644);
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <bitset>
|
||||
#include <cassert>
|
||||
#include <cstring>
|
||||
#include <fcntl.h>
|
||||
#include <fstream>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/user.h>
|
||||
@@ -363,7 +364,10 @@ namespace FEX::HarnessHelper {
|
||||
public:
|
||||
|
||||
HarnessCodeLoader(std::string const &Filename, const char *ConfigFilename) {
|
||||
ReadFile(Filename, &RawFile);
|
||||
TestFD = open(Filename.c_str(), O_CLOEXEC | O_RDONLY);
|
||||
TestFileSize = lseek(TestFD, 0, SEEK_END);
|
||||
lseek(TestFD, 0, SEEK_SET);
|
||||
|
||||
if (ConfigFilename) {
|
||||
Config.Init(ConfigFilename);
|
||||
}
|
||||
@@ -390,8 +394,8 @@ namespace FEX::HarnessHelper {
|
||||
|
||||
bool MapMemory(const MapperFn& Mapper, const UnmapperFn& Unmapper) override {
|
||||
bool LimitedSize = true;
|
||||
auto DoMMap = [](uint64_t Address, size_t Size) -> void* {
|
||||
void *Result = FEXCore::Allocator::mmap(reinterpret_cast<void*>(Address), Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
auto DoMMap = [&Mapper](uint64_t Address, size_t Size) -> void* {
|
||||
void *Result = Mapper(reinterpret_cast<void*>(Address), Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
LOGMAN_THROW_A_FMT(Result == reinterpret_cast<void*>(Address), "Map Memory mmap failed");
|
||||
return Result;
|
||||
};
|
||||
@@ -414,9 +418,8 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
|
||||
// Map in the memory region for the test file
|
||||
size_t Length = FEXCore::AlignUp(RawFile.size(), FHU::FEX_PAGE_SIZE);
|
||||
Code_start_page = reinterpret_cast<uint64_t>(DoMMap(Code_start_page, Length));
|
||||
mprotect(reinterpret_cast<void*>(Code_start_page), Length, PROT_READ | PROT_WRITE | PROT_EXEC);
|
||||
size_t Length = FEXCore::AlignUp(TestFileSize, FHU::FEX_PAGE_SIZE);
|
||||
Mapper(reinterpret_cast<void*>(Code_start_page), Length, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_FIXED | MAP_PRIVATE, TestFD, 0);
|
||||
RIP = Code_start_page;
|
||||
|
||||
// Map the memory regions the test file asks for
|
||||
@@ -432,7 +435,6 @@ namespace FEX::HarnessHelper {
|
||||
void LoadMemory() {
|
||||
// Memory base here starts at the start location we passed back with GetLayout()
|
||||
// This will write at [CODE_START_RANGE + 0, RawFile.size() )
|
||||
memcpy(reinterpret_cast<void*>(RIP), &RawFile.at(0), RawFile.size());
|
||||
Config.LoadMemory();
|
||||
}
|
||||
|
||||
@@ -453,7 +455,8 @@ namespace FEX::HarnessHelper {
|
||||
uint64_t Code_start_page = 0x1'0000;
|
||||
uint64_t RIP {};
|
||||
|
||||
std::vector<char> RawFile;
|
||||
int TestFD{};
|
||||
size_t TestFileSize{};
|
||||
ConfigLoader Config;
|
||||
};
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ $end_info$
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
@@ -119,6 +120,31 @@ private:
|
||||
constexpr static uint64_t STACK_SIZE = 8 * 1024 * 1024;
|
||||
};
|
||||
|
||||
class DummySyscallHandler: public FEXCore::HLE::SyscallHandler {
|
||||
public:
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) override {
|
||||
LOGMAN_MSG_A_FMT("Syscalls not implemented");
|
||||
return 0;
|
||||
}
|
||||
|
||||
FEXCore::HLE::SyscallABI GetSyscallABI(uint64_t Syscall) override {
|
||||
LOGMAN_MSG_A_FMT("Syscalls not implemented");
|
||||
return {0, false, 0 };
|
||||
}
|
||||
|
||||
// These are no-ops implementations of the SyscallHandler API
|
||||
std::shared_mutex StubMutex;
|
||||
FEXCore::HLE::AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(uint64_t GuestAddr) override {
|
||||
return {0, 0, FHU::ScopedSignalMaskWithSharedLock {StubMutex}};
|
||||
}
|
||||
|
||||
std::shared_mutex StubMutex2;
|
||||
std::shared_lock<std::shared_mutex> CompileCodeLock(uint64_t Start) {
|
||||
return std::shared_lock(StubMutex2);
|
||||
}
|
||||
};
|
||||
|
||||
int main(int argc, char **argv, char **const envp)
|
||||
{
|
||||
LogMan::Throw::InstallHandler(AssertHandler);
|
||||
@@ -141,6 +167,7 @@ int main(int argc, char **argv, char **const envp)
|
||||
std::unique_ptr<FEX::HLE::SignalDelegator> SignalDelegation = std::make_unique<FEX::HLE::SignalDelegator>();
|
||||
|
||||
FEXCore::Context::SetSignalDelegator(CTX, SignalDelegation.get());
|
||||
FEXCore::Context::SetSyscallHandler(CTX, new DummySyscallHandler());
|
||||
|
||||
FEX::IRLoader::Loader Loader(Args[0], Args[1]);
|
||||
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
#include "IRLoader/Loader.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
@@ -17,12 +19,12 @@ namespace FEX::IRLoader {
|
||||
return;
|
||||
}
|
||||
|
||||
ParsedCode = FEXCore::IR::Parse(&fp);
|
||||
ParsedCode = FEXCore::IR::Parse(Allocator, &fp);
|
||||
|
||||
if (ParsedCode) {
|
||||
auto NewIR = ParsedCode->ViewIR();
|
||||
EntryRIP = 0x40000;
|
||||
|
||||
|
||||
std::stringstream out;
|
||||
FEXCore::IR::Dump(&out, &NewIR, nullptr);
|
||||
fmt::print("IR:\n{}\n@@@@@\n", out.str());
|
||||
|
||||
@@ -41,6 +41,7 @@ namespace FEX::IRLoader {
|
||||
uint64_t EntryRIP{};
|
||||
std::unique_ptr<IREmitter> ParsedCode;
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMalloc Allocator;
|
||||
FEX::HarnessHelper::ConfigLoader Config;
|
||||
};
|
||||
}
|
||||
@@ -335,6 +335,7 @@ enum Syscalls_Arm64 {
|
||||
SYSCALL_Arm64_memfd_secret = 447,
|
||||
SYSCALL_Arm64_process_mrelease = 448,
|
||||
SYSCALL_Arm64_futex_waitv = 449,
|
||||
SYSCALL_Arm64_set_mempolicy_home_node = 450,
|
||||
SYSCALL_Arm64_MAX = 512,
|
||||
|
||||
// Unsupported syscalls on this host
|
||||
|
||||
@@ -6,6 +6,8 @@ add_library(LinuxEmulation STATIC
|
||||
LinuxAllocator.cpp
|
||||
SignalDelegator.cpp
|
||||
Syscalls.cpp
|
||||
SyscallsSMCTracking.cpp
|
||||
SyscallsVMATracking.cpp
|
||||
x32/Syscalls.cpp
|
||||
x32/EPoll.cpp
|
||||
x32/FD.cpp
|
||||
@@ -69,6 +71,7 @@ PRIVATE
|
||||
-Werror=implicit-fallthrough
|
||||
|
||||
-Wno-trigraphs
|
||||
-fwrapv
|
||||
)
|
||||
|
||||
target_include_directories(LinuxEmulation
|
||||
@@ -101,7 +104,8 @@ math(EXPR ARG_COUNT "${ARG_COUNT}-1")
|
||||
set (ARGS
|
||||
"-x" "c++"
|
||||
"-std=c++20"
|
||||
"-fno-operator-names")
|
||||
"-fno-operator-names"
|
||||
"-I${PROJECT_SOURCE_DIR}/External/drm-headers/include/")
|
||||
# Global include directories
|
||||
get_directory_property (INC_DIRS INCLUDE_DIRECTORIES)
|
||||
list(TRANSFORM INC_DIRS PREPEND "-I")
|
||||
|
||||
@@ -45,7 +45,7 @@ namespace FEX::EmulatedFile {
|
||||
* @return A temporary file that we can use
|
||||
*/
|
||||
static int GenTmpFD() {
|
||||
int fd = open("/tmp", O_RDWR | O_TMPFILE | O_EXCL | S_IRUSR | S_IWUSR);
|
||||
int fd = open("/tmp", O_RDWR | O_TMPFILE | O_EXCL, S_IRUSR | S_IWUSR);
|
||||
return fd;
|
||||
}
|
||||
|
||||
|
||||
Loaded 100 of 267 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user