mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 06:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cc85a6a722 | ||
|
|
e72fa02897 | ||
|
|
8a4c5bcc65 | ||
|
|
7a13a24c05 | ||
|
|
5a53931b92 | ||
|
|
f9b352a093 | ||
|
|
f444b03317 | ||
|
|
ed05846dd0 | ||
|
|
f609990f90 | ||
|
|
d2032da452 | ||
|
|
8047007a7a | ||
|
|
1f7e82ea09 | ||
|
|
35c52f20f9 | ||
|
|
17c82c22a6 | ||
|
|
df03a7b101 | ||
|
|
c859540d7e | ||
|
|
20794593e7 | ||
|
|
a80a2bf569 | ||
|
|
e86a792189 | ||
|
|
d6c9b549df | ||
|
|
1a4d5a1abb | ||
|
|
c3e123df25 | ||
|
|
cac798574a | ||
|
|
1506a19229 | ||
|
|
51861234bc | ||
|
|
677b72c9a5 | ||
|
|
71a8c66c95 | ||
|
|
3372e9bdbb | ||
|
|
df0723e14b | ||
|
|
7ee6fc0d7f | ||
|
|
bf773452ac | ||
|
|
95dbccc0ab | ||
|
|
e5189d63a2 | ||
|
|
7d5442357a | ||
|
|
9dcc1deec0 | ||
|
|
66d4206cd7 | ||
|
|
f39163b1e1 | ||
|
|
01837b3ad6 | ||
|
|
bdb68840e3 | ||
|
|
4e2dcf3298 | ||
|
|
628f825416 | ||
|
|
a80327f6df | ||
|
|
16f7002222 | ||
|
|
c9712e45cb | ||
|
|
a082161d72 | ||
|
|
9017325c95 | ||
|
|
ae536e44d7 | ||
|
|
7679485cc3 | ||
|
|
cb8bf1add6 | ||
|
|
a69c457715 | ||
|
|
7c4729678b | ||
|
|
537562fab7 | ||
|
|
f8721992c2 | ||
|
|
e652399fd3 | ||
|
|
e7c92c43a0 | ||
|
|
9b5e1c44c8 | ||
|
|
755600c371 | ||
|
|
bec8b70e5d | ||
|
|
bef8ddde48 | ||
|
|
fe06f1b151 | ||
|
|
92a15e00c7 | ||
|
|
2997257d6d | ||
|
|
7ceadc6b5b | ||
|
|
8c41e8f7d8 | ||
|
|
784b3064fc |
No files matched your search
@@ -7,6 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
@@ -179,8 +180,13 @@ endif()
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
set(BUILD_FEXCONFIG FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
|
||||
@@ -282,6 +282,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tIROps Op;\n\n")
|
||||
output_file.write("\tuint8_t Size;\n")
|
||||
output_file.write("\tuint8_t ElementSize;\n")
|
||||
output_file.write("\tuint8_t _pad;\n")
|
||||
|
||||
output_file.write("\ttemplate<typename T>\n")
|
||||
output_file.write("\tT const* C() const { return reinterpret_cast<T const*>(Data); }\n")
|
||||
@@ -291,6 +292,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tOrderedNodeWrapper Args[0];\n")
|
||||
|
||||
output_file.write("};\n\n");
|
||||
output_file.write("static_assert(sizeof(IROp_Header) == sizeof(uint32_t), \"IROp_Header should be 32-bits in size\");\n\n");
|
||||
|
||||
# Now the user defined types
|
||||
output_file.write("// User defined IR Op structs\n")
|
||||
|
||||
+1
-1
@@ -231,7 +231,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl xxhash tiny-json FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
|
||||
+5
-1
@@ -17,8 +17,12 @@ namespace FEXCore {
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
#endif
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
|
||||
-216
@@ -27,8 +27,6 @@
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
}
|
||||
@@ -42,71 +40,6 @@ namespace DefaultValues {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
@@ -583,154 +516,5 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -66,10 +66,6 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::AddVirtualMemoryMapping([[maybe_unused]] uint64_t VirtualAddress, [[maybe_unused]] uint64_t PhysicalAddress, [[maybe_unused]] uint64_t Size) {
|
||||
return false;
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
+6
-7
@@ -100,8 +100,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
@@ -154,7 +152,9 @@ namespace FEXCore::Context {
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
@@ -254,7 +254,7 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
@@ -291,7 +291,7 @@ namespace FEXCore::Context {
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -302,8 +302,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
ScopedDeferredSignalWithUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
+196
-76
@@ -20,6 +20,131 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
}
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
{FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
@@ -29,22 +154,21 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
ConfiguredGPRs = NumGPRs64;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs64;
|
||||
ConfiguredGPRPairs = NumGPRPairs64;
|
||||
ConfiguredFPRs = NumFPRs64;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs64;
|
||||
ConfiguredDynamicGPRs = NumGPRs64 - NumGPRs64; // Will be zero, just to be consistent with 32-bit side
|
||||
ConfiguredDynamicRegisterBase = nullptr;
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredGPRs = NumGPRs32;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs32;
|
||||
ConfiguredGPRPairs = NumGPRPairs32;
|
||||
ConfiguredFPRs = NumFPRs32;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs32;
|
||||
ConfiguredDynamicGPRs = NumGPRs32 - NumGPRs64; // Will be 8
|
||||
ConfiguredDynamicRegisterBase = &RA64[9];
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralPairRegisters = x32::RAPair;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,9 +348,9 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -241,8 +365,8 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
@@ -254,18 +378,18 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
@@ -297,8 +421,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TMP4.R());
|
||||
@@ -308,22 +432,22 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRFillMask)];
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
@@ -340,9 +464,9 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -358,10 +482,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicGPRs + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = ConfiguredFPRs * FPRRegSize;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -370,31 +494,29 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(ConfiguredFPRs % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
@@ -404,30 +526,28 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
@@ -26,87 +26,9 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9 + 8> RA64 = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4 + 3> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
// Registers only available on 32-bit
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17}
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12 + 8> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
@@ -139,32 +61,16 @@ protected:
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
uint32_t ConfiguredGPRs;
|
||||
uint32_t ConfiguredSRAGPRs;
|
||||
uint32_t ConfiguredGPRPairs;
|
||||
uint32_t ConfiguredFPRs;
|
||||
uint32_t ConfiguredSRAFPRs;
|
||||
uint32_t ConfiguredDynamicGPRs;
|
||||
const FEXCore::ARMEmitter::Register *ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
// 64-bit gets removal of additional pairs
|
||||
constexpr static uint32_t NumGPRs64 = RA64.size() - 8;
|
||||
constexpr static uint32_t NumSRAGPRs64 = SRA64.size();
|
||||
constexpr static uint32_t NumFPRs64 = RAFPR.size() - 8;
|
||||
constexpr static uint32_t NumSRAFPRs64 = SRAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs64 = RA64Pair.size() - 3;
|
||||
|
||||
// 32-bit gets full array of GPR registers
|
||||
// SRA registers remove the additional 8
|
||||
constexpr static uint32_t NumGPRs32 = RA64.size();
|
||||
constexpr static uint32_t NumSRAGPRs32 = SRA64.size() - 8;
|
||||
constexpr static uint32_t NumFPRs32 = RAFPR.size();
|
||||
constexpr static uint32_t NumSRAFPRs32 = SRAFPR.size() - 8;
|
||||
constexpr static uint32_t NumGPRPairs32 = RA64Pair.size();
|
||||
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
|
||||
@@ -145,6 +145,27 @@ public:
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0000, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0001, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0010, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0011, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
@@ -381,6 +402,26 @@ public:
|
||||
(0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
@@ -467,7 +508,24 @@ public:
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
void ctz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
// TODO: PAUTH
|
||||
|
||||
@@ -819,6 +877,21 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void MinMaxImmediate(uint32_t opc, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = 0b1'0001'11U << 22;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= opc << 18;
|
||||
Instr |= (Imm & 0xFF) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Move Wide
|
||||
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -899,6 +972,9 @@ private:
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == FEXCore::ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
|
||||
+3
-5
@@ -395,8 +395,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -429,14 +427,14 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(CTX->HostFeatures.SupportsCRC << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
(0 << 24) | // APIC TSC-Deadline
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 26) | // XSAVE
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
|
||||
+44
-9
@@ -8,6 +8,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
@@ -24,6 +25,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -195,17 +197,35 @@ namespace FEXCore::Context {
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
const CPU::CPUBackend::JITCodeHeader *InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
const CPU::CPUBackend::JITCodeTail *InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
|
||||
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
return InlineTail->RIP;
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto &RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
}
|
||||
else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
}
|
||||
}
|
||||
return StartingGuestRIP;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -645,7 +665,17 @@ namespace FEXCore::Context {
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::CleanupAfterFork(FEXCore::Core::InternalThreadState *LiveThread) {
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState *LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto &DeadThread : Threads) {
|
||||
@@ -684,6 +714,11 @@ namespace FEXCore::Context {
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
}
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
@@ -811,7 +846,7 @@ namespace FEXCore::Context {
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
if (ExtendedDebugInfo) {
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
|
||||
@@ -1030,7 +1065,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
std::shared_lock lk(CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1210,7 +1245,7 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread);
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
@@ -1219,7 +1254,7 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread);
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1247,7 +1282,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::shared_lock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
@@ -31,22 +31,22 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
void EmitDispatcher();
|
||||
|
||||
uint16_t GetSRAGPRCount() const override {
|
||||
return SRA64.size();
|
||||
return StaticRegisters.size();
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const override {
|
||||
return SRAFPR.size();
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRA64.size(); ++i) {
|
||||
Mapping[i] = SRA64[i].Idx();
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRAFPR.size(); ++i) {
|
||||
Mapping[i] = SRAFPR[i].Idx();
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -46,7 +46,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const uint32_t upper_limit = (16U >> (control & 1)) - 1;
|
||||
const int32_t upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
|
||||
@@ -2280,16 +2280,13 @@ DEF_OP(VRev64) {
|
||||
|
||||
DEF_OP(VPCMPESTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto RAX = *GetSrc<uint64_t*>(Data->SSAData, Op->RAX);
|
||||
const auto RDX = *GetSrc<uint64_t*>(Data->SSAData, Op->RDX);
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPESTRX>::handle(RAX, RDX, LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
|
||||
@@ -477,9 +477,9 @@ DEF_OP(PDep) {
|
||||
const auto IndexReg = TMP4.R();
|
||||
const auto ZeroReg = ARMEmitter::Reg::zr;
|
||||
|
||||
const auto InputReg = SRA64[0];
|
||||
const auto MaskReg = SRA64[1];
|
||||
const auto DestReg = SRA64[2];
|
||||
const auto InputReg = StaticRegisters[0];
|
||||
const auto MaskReg = StaticRegisters[1];
|
||||
const auto DestReg = StaticRegisters[2];
|
||||
|
||||
const auto SpillCode = 1U << InputReg.Idx() |
|
||||
1U << MaskReg.Idx() |
|
||||
|
||||
+28
-10
@@ -450,16 +450,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
|
||||
@@ -587,14 +584,14 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, ConfiguredGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, ConfiguredSRAGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, ConfiguredFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, ConfiguredSRAFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, ConfiguredGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, GeneralPairRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < ConfiguredGPRPairs; ++i) {
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
@@ -1105,12 +1102,33 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
@@ -70,9 +70,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -84,9 +84,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -97,7 +97,7 @@ private:
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return RA64Pair[Reg.Reg];
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
+119
-51
@@ -132,7 +132,7 @@ DEF_OP(LoadRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -166,7 +166,7 @@ DEF_OP(LoadRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
@@ -214,7 +214,7 @@ DEF_OP(StoreRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
@@ -248,7 +248,7 @@ DEF_OP(StoreRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
@@ -296,9 +296,9 @@ DEF_OP(LoadRegisterSRA) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -334,9 +334,9 @@ DEF_OP(LoadRegisterSRA) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -484,9 +484,9 @@ DEF_OP(StoreRegisterSRA) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -520,9 +520,9 @@ DEF_OP(StoreRegisterSRA) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -1260,7 +1260,6 @@ DEF_OP(LoadMemTSO) {
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -2023,32 +2022,78 @@ DEF_OP(MemCpy) {
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldapr(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldapr(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(Dst, Addr);
|
||||
ldarb(Dst, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
ldarh(Dst, Addr);
|
||||
ldarh(Dst, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldar(Dst.W(), Addr);
|
||||
ldar(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldar(Dst.X(), Addr);
|
||||
ldar(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
@@ -2059,31 +2104,30 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i8Bit, Dst, 0, TMP1);
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
ldarh(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 0, TMP1);
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
ldar(TMP1.W(), Addr);
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 8:
|
||||
ldar(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case 16:
|
||||
nop();
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, Addr);
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), Addr);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default:
|
||||
@@ -2097,26 +2141,50 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
stlurh(Src, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
stlur(Src.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
stlur(Src.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
stlrb(Src, Addr);
|
||||
stlrb(Src, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
stlrh(Src, Addr);
|
||||
stlrh(Src, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
stlr(Src.W(), Addr);
|
||||
stlr(Src.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
stlr(Src.X(), Addr);
|
||||
stlr(Src.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
@@ -2129,19 +2197,19 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, Addr);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, Addr);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), Addr);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, Addr);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case 16: {
|
||||
// Move vector to GPRs
|
||||
@@ -2151,14 +2219,14 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, Addr); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, Addr); // <- Can also hit SIGBUS
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, Addr, 0);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
|
||||
+22
-4
@@ -308,16 +308,13 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto SrcRAX = GetSrc<RA_64>(Op->RAX.ID());
|
||||
const auto SrcRDX = GetSrc<RA_64>(Op->RDX.ID());
|
||||
|
||||
// Encode the size check into the 8th bit to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(rdi, SrcRAX);
|
||||
mov(rsi, SrcRDX);
|
||||
|
||||
@@ -820,12 +817,33 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
auto JITBlockTail = getCurr<JITCodeTail*>();
|
||||
setSize(getSize() + sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = getCurr<uint8_t *>();
|
||||
auto JITRIPEntries = getCurr<JITRIPReconstructEntries*>();
|
||||
|
||||
setSize(getSize() + sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = getCurr<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
+13
-8
@@ -5477,11 +5477,6 @@ void OpDispatchBuilder::MOVBEOp(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void OpDispatchBuilder::FenceOp(OpcodeArgs) {
|
||||
_Fence({FenceType});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CLWB(OpcodeArgs) {
|
||||
OrderedNode *DestMem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
DestMem = AppendSegmentOffset(DestMem, Op->Flags);
|
||||
@@ -5494,6 +5489,15 @@ void OpDispatchBuilder::CLFLUSHOPT(OpcodeArgs) {
|
||||
_CacheLineClear(DestMem, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LoadFenceOrXRSTOR(OpcodeArgs) {
|
||||
// 0xE8 signifies LFENCE
|
||||
if (Op->ModRM == 0xE8) {
|
||||
_Fence(IR::Fence_Load);
|
||||
} else {
|
||||
XRstorOpImpl(Op);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MemFenceOrXSAVEOPT(OpcodeArgs) {
|
||||
if (Op->ModRM == 0xF0) {
|
||||
// 0xF0 is MFENCE
|
||||
@@ -6713,9 +6717,10 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 1), 1, &OpDispatchBuilder::FXRStoreOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 2), 1, &OpDispatchBuilder::LDMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 3), 1, &OpDispatchBuilder::STMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::FenceOp<FEXCore::IR::Fence_Load.Val>}, //LFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, //MFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, //SFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 4), 1, &OpDispatchBuilder::XSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::LoadFenceOrXRSTOR}, // LFENCE (or XRSTOR)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, // MFENCE (or XSAVEOPT)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, // SFENCE (or CLFLUSH)
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
+45
-3
@@ -149,6 +149,31 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool CanHaveSideEffects(FEXCore::X86Tables::X86InstInfo const* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
|
||||
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
}
|
||||
|
||||
auto CanHaveSideEffects = false;
|
||||
|
||||
auto HasPotentialMemoryAccess = [](X86Tables::DecodedOperand const &Operand) -> bool {
|
||||
if (Operand.IsNone()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// This isn't guaranteed that all of these types will access memory, but be safe.
|
||||
return Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB();
|
||||
};
|
||||
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Dest);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[0]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[1]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[2]);
|
||||
return CanHaveSideEffects;
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
|
||||
@@ -699,6 +724,8 @@ public:
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
void XSaveOp(OpcodeArgs);
|
||||
|
||||
void PAlignrOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void UCOMISxOp(OpcodeArgs);
|
||||
@@ -748,11 +775,9 @@ public:
|
||||
void PHADDS(OpcodeArgs);
|
||||
void PHSUBS(OpcodeArgs);
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void FenceOp(OpcodeArgs);
|
||||
|
||||
void CLWB(OpcodeArgs);
|
||||
void CLFLUSHOPT(OpcodeArgs);
|
||||
void LoadFenceOrXRSTOR(OpcodeArgs);
|
||||
void MemFenceOrXSAVEOPT(OpcodeArgs);
|
||||
void StoreFenceOrCLFlush(OpcodeArgs);
|
||||
void CLZeroOp(OpcodeArgs);
|
||||
@@ -953,6 +978,23 @@ private:
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
|
||||
void XSaveOpImpl(OpcodeArgs);
|
||||
void SaveX87State(OpcodeArgs, OrderedNode *MemBase);
|
||||
void SaveSSEState(OrderedNode *MemBase);
|
||||
void SaveMXCSRState(OrderedNode *MemBase);
|
||||
void SaveAVXState(OrderedNode *MemBase);
|
||||
|
||||
void XRstorOpImpl(OpcodeArgs);
|
||||
void RestoreX87State(OrderedNode *MemBase);
|
||||
void RestoreSSEState(OrderedNode *MemBase);
|
||||
void RestoreMXCSRState(OrderedNode *MXCSR);
|
||||
void RestoreAVXState(OrderedNode *MemBase);
|
||||
void DefaultX87State(OpcodeArgs);
|
||||
void DefaultSSEState();
|
||||
void DefaultAVXState();
|
||||
|
||||
OrderedNode *GetMXCSR();
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
|
||||
+256
-26
@@ -2455,6 +2455,76 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
SaveX87State(Op, Mem);
|
||||
SaveSSEState(Mem);
|
||||
SaveMXCSRState(Mem);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOp(OpcodeArgs) {
|
||||
XSaveOpImpl(Op);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// NOTE: Mask should be EAX and EDX concatenated, but we only need to test
|
||||
// for features that are in the lower 32 bits, so EAX only is sufficient.
|
||||
OrderedNode *Mask = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *Base = XSaveBase();
|
||||
|
||||
const auto StoreIfFlagSet = [&](uint32_t BitIndex, auto fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
fn();
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetFalseJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
StoreIfFlagSet(0, [this, Op, Base] { SaveX87State(Op, Base); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveSSEState(Base); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
StoreIfFlagSet(2, [this, Base] { SaveAVXState(Base); });
|
||||
}
|
||||
|
||||
// We need to save MXCSR and MXCSR_MASK if either SSE or AVX are requested to be saved
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveMXCSRState(Base); }, 2);
|
||||
}
|
||||
|
||||
// Update XSTATE_BV region of the XSAVE header
|
||||
{
|
||||
OrderedNode *HeaderOffset = _Add(Base, _Constant(512));
|
||||
|
||||
// NOTE: We currently only support the first 3 bits (x87, SSE, and AVX)
|
||||
OrderedNode *RequestedFeatures = _Bfe(3, 0, Mask);
|
||||
|
||||
// XSTATE_BV section of the header is 8 bytes in size, but we only really
|
||||
// care about setting at most 3 bits in the first byte. We zero out the rest.
|
||||
_StoreMem(GPRClass, 8, HeaderOffset, RequestedFeatures);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveX87State(OpcodeArgs, OrderedNode *MemBase) {
|
||||
// Saves 512bytes to the memory location provided
|
||||
// Header changes depending on if REX.W is set or not
|
||||
if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) {
|
||||
@@ -2472,12 +2542,12 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, 2, Mem, FCW, 2);
|
||||
_StoreMem(GPRClass, 2, MemBase, FCW, 2);
|
||||
}
|
||||
|
||||
{
|
||||
// We must construct the FSW from our various bits
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
OrderedNode *FSW = _Constant(0);
|
||||
auto Top = GetX87Top();
|
||||
FSW = _Or(FSW, _Lshl(Top, _Constant(11)));
|
||||
@@ -2496,7 +2566,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
_StoreMem(GPRClass, 2, MemLocation, FTW, 2);
|
||||
}
|
||||
@@ -2545,33 +2615,142 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, MMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(OrderedNode *MemBase) {
|
||||
OrderedNode *MXCSR = GetMXCSR();
|
||||
OrderedNode *MXCSRLocation = _Add(MemBase, _Constant(24));
|
||||
_StoreMem(GPRClass, 4, MXCSRLocation, MXCSR, 4);
|
||||
|
||||
// Store the mask for all bits.
|
||||
OrderedNode *MXCSRMaskLocation = _Add(MXCSRLocation, _Constant(4));
|
||||
_StoreMem(GPRClass, 4, MXCSRMaskLocation, _Constant(0xFFFF), 4);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *Upper = _VDupElement(32, 16, LoadXMMRegister(i), 1);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, Upper, 16);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetMXCSR() {
|
||||
// Default MXCSR Value
|
||||
OrderedNode *MXCSR = _Constant(0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
return _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
RestoreX87State(Mem);
|
||||
RestoreSSEState(Mem);
|
||||
|
||||
OrderedNode *MXCSRLocation = _Add(Mem, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// Set up base address for the XSAVE region to restore from, and also read the
|
||||
// XSTATE_BV bit flags out of the XSTATE header.
|
||||
OrderedNode *Base = XSaveBase();
|
||||
OrderedNode *Mask = _LoadMem(GPRClass, 8, _Add(Base, _Constant(512)), 8);
|
||||
|
||||
// If a bit in our XSTATE_BV is set, then we restore from that region of the XSAVE area,
|
||||
// otherwise, if not set, then we need to set the relevant data the bit corresponds to
|
||||
// to it's defined initial configuration.
|
||||
const auto RestoreIfFlagSetOrDefault = [&](uint32_t BitIndex, auto restore_fn, auto default_fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto RestoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, RestoreBlock);
|
||||
SetCurrentCodeBlock(RestoreBlock);
|
||||
{
|
||||
restore_fn();
|
||||
}
|
||||
auto RestoreExitJump = _Jump();
|
||||
auto DefaultBlock = CreateNewCodeBlockAfter(RestoreBlock);
|
||||
auto ExitBlock = CreateNewCodeBlockAfter(DefaultBlock);
|
||||
SetJumpTarget(RestoreExitJump, ExitBlock);
|
||||
SetFalseJumpTarget(CondJump, DefaultBlock);
|
||||
SetCurrentCodeBlock(DefaultBlock);
|
||||
{
|
||||
default_fn();
|
||||
}
|
||||
auto DefaultExitJump = _Jump();
|
||||
SetJumpTarget(DefaultExitJump, ExitBlock);
|
||||
SetCurrentCodeBlock(ExitBlock);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(0,
|
||||
[this, Base] { RestoreX87State(Base); },
|
||||
[this, Op] { DefaultX87State(Op); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] { RestoreSSEState(Base); },
|
||||
[this] { DefaultSSEState(); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(2,
|
||||
[this, Base] { RestoreAVXState(Base); },
|
||||
[this] { DefaultAVXState(); });
|
||||
}
|
||||
|
||||
{
|
||||
// We need to restore the MXCSR if either SSE or AVX are requested to be saved
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] {
|
||||
OrderedNode *MXCSRLocation = _Add(Base, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
},
|
||||
[] { /* Intentionally do nothing*/ }, 2);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreX87State(OrderedNode *MemBase) {
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, MemBase, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
{
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
|
||||
// Strip out the FSW information
|
||||
@@ -2591,26 +2770,78 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto NewFTW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
}
|
||||
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreMXCSRState(OrderedNode *MXCSR) {
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, MXCSR);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
OrderedNode *YMMHReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
OrderedNode *YMM = _VInsElement(32, 16, 1, 0, XMMReg, YMMHReg);
|
||||
StoreXMMRegister(i, YMM);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
|
||||
// We can piggy-back on FNINIT's implementation, since
|
||||
// it performs the same behavior as required by XRSTOR for resetting flags
|
||||
FNINIT(Op);
|
||||
|
||||
// On top of resetting the flags to a default state, we also need to clear
|
||||
// all of the ST0-7/MM0-7 registers to zero.
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::MM_REG_SIZE);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
_StoreContext(16, FPRClass, ZeroVector, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultSSEState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
StoreXMMRegister(i, ZeroVector);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultAVXState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
OrderedNode* Reg = LoadXMMRegister(i);
|
||||
OrderedNode* Dst = _VMov(16, Reg);
|
||||
StoreXMMRegister(i, Dst);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
@@ -2676,18 +2907,11 @@ void OpDispatchBuilder::UCOMISxOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::LDMXCSR(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, Dest);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
RestoreMXCSRState(Dest);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::STMXCSR(OpcodeArgs) {
|
||||
// Default MXCSR
|
||||
OrderedNode *MXCSR = _Constant(32, 0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
MXCSR = _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
|
||||
StoreResult(GPRClass, Op, MXCSR, -1);
|
||||
StoreResult(GPRClass, Op, GetMXCSR(), -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PACKUSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
@@ -4593,7 +4817,7 @@ void OpDispatchBuilder::VPERMILRegOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[1] needs to be a literal");
|
||||
const auto Control = Op->Src[1].Data.Literal.Value;
|
||||
const uint16_t Control = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
// SSE4.2 string instructions modify flags, so invalidate
|
||||
// any previously deferred flags.
|
||||
@@ -4611,12 +4835,18 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
OrderedNode *IntermediateResult{};
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
//
|
||||
// While the control bit immediate for the instruction itself is only ever 8 bits
|
||||
// in size, we use it as a 16-bit value so that we can use the 8th bit to signify
|
||||
// whether or not RAX and RDX should be interpreted as a 64-bit value.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is64Bit = SrcSize == 8;
|
||||
const auto NewControl = uint16_t(Control | (uint16_t(Is64Bit) << 8));
|
||||
|
||||
OrderedNode *SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(SrcSize, Src1, Src2, SrcRAX, SrcRDX, Control);
|
||||
IntermediateResult = _VPCMPESTRX(Src1, Src2, SrcRAX, SrcRDX, NewControl);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
|
||||
@@ -146,10 +146,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
|
||||
@@ -334,7 +334,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 1), 1, X86InstInfo{"FXRSTOR", TYPE_INST, FLAGS_MODRM, 0, nullptr}}, // MMX/x87
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, X86InstInfo{"LDMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, X86InstInfo{"STMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 5), 1, X86InstInfo{"LFENCE/XRSTOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 6), 1, X86InstInfo{"MFENCE/XSAVEOPT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 7), 1, X86InstInfo{"SFENCE/CLFLUSH", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
+1
-3
@@ -1405,7 +1405,7 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"GPR = VPCMPESTRX u8:$GPRSize, FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u8:$Control": {
|
||||
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
@@ -1414,7 +1414,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
@@ -1426,7 +1425,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
|
||||
+12
@@ -340,5 +340,17 @@ namespace FEXCore::Allocator {
|
||||
::munmap(Region.Ptr, Region.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Alloc64) {
|
||||
Alloc64->LockBeforeFork(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {
|
||||
if (Alloc64) {
|
||||
Alloc64->UnlockAfterFork(Thread, Child);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child);
|
||||
}
|
||||
+17
-4
@@ -49,6 +49,19 @@ namespace Alloc::OSAllocator {
|
||||
void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) override;
|
||||
int Munmap(void *addr, size_t length) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override {
|
||||
AllocationMutex.lock();
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override {
|
||||
if (Child) {
|
||||
AllocationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
AllocationMutex.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Upper bound is the maximum virtual address space of the host processor
|
||||
uintptr_t UPPER_BOUND = (1ULL << 57);
|
||||
@@ -139,7 +152,7 @@ namespace Alloc::OSAllocator {
|
||||
LiveRegionListType *LiveRegions{};
|
||||
|
||||
Alloc::ForwardOnlyIntrusiveArenaAllocator *ObjectAlloc{};
|
||||
std::mutex AllocationMutex{};
|
||||
FEXCore::ForkableUniqueMutex AllocationMutex;
|
||||
void DetermineVASize();
|
||||
|
||||
LiveVMARegion *MakeRegionActive(ReservedRegionListType::iterator ReservedIterator, uint64_t UsedSize) {
|
||||
@@ -258,7 +271,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -446,7 +459,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -571,7 +584,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
@@ -22,6 +22,9 @@ namespace Alloc {
|
||||
|
||||
virtual void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) { return nullptr; }
|
||||
virtual int Munmap(void *addr, size_t length) { return -1; }
|
||||
|
||||
virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
};
|
||||
|
||||
class GlobalAllocator {
|
||||
|
||||
+3
-3
@@ -2043,8 +2043,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
@@ -2068,7 +2068,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
return std::make_pair(true, -4);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
|
||||
@@ -244,50 +244,4 @@ namespace Type {
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
};
|
||||
|
||||
// Application loaders
|
||||
class FEX_DEFAULT_VISIBILITY OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit OptionMapper(FEXCore::Config::LayerType Layer);
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char *ConfigName, const char *ConfigString);
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
|
||||
/**
|
||||
* @brief Loads the main application config
|
||||
*
|
||||
* @param File Optional override to load a specific config file in to the main layer
|
||||
* Shouldn't be commonly used
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
*
|
||||
* @param Filename Application filename component
|
||||
* @param Global Load the global configuration or user accessible file
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
*
|
||||
* @param _envp[] The environment array from main
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]);
|
||||
|
||||
}
|
||||
@@ -89,6 +89,28 @@ namespace CPU {
|
||||
size_t Size;
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
// Offset after this block to the start of the RIP entries.
|
||||
uint32_t OffsetToRIPEntries;
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using 16-bit entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
struct JITRIPReconstructEntries {
|
||||
// The Host PC offset from the previous entry.
|
||||
uint16_t HostPCOffset;
|
||||
|
||||
// How much to offset the RIP from the previous entry.
|
||||
uint16_t GuestRIPOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
+2
-12
@@ -225,17 +225,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) = 0;
|
||||
|
||||
/**
|
||||
* @brief Sets up memory regions on the guest for mirroring within the guest's VM space
|
||||
*
|
||||
* @param VirtualAddress The address we want to set to mirror a physical memory region
|
||||
* @param PhysicalAddress The physical memory region we are mapping
|
||||
* @param Size Size of the region to mirror
|
||||
*
|
||||
* @return true when successfully mapped. false if there was an error adding
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Retrieves a feature struct indicating certain supported aspects from
|
||||
* the hose.
|
||||
@@ -254,7 +243,8 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void StopThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) = 0;
|
||||
|
||||
|
||||
+29
-5
@@ -332,8 +332,10 @@ static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
|
||||
struct RegisterClassType final {
|
||||
uint32_t Val;
|
||||
[[nodiscard]] constexpr operator uint32_t() const {
|
||||
using value_type = uint32_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]] friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
@@ -388,8 +390,10 @@ struct TypeDefinition final {
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
using value_type = uint8_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]] friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
|
||||
@@ -602,7 +606,27 @@ struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID:
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) {
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
|
||||
return Base::format(ID.Value, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
|
||||
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
|
||||
return Base::format(Class.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
|
||||
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
|
||||
return Base::format(Fence.Val, ctx);
|
||||
}
|
||||
};
|
||||
+8
-1
@@ -112,7 +112,14 @@ namespace FEXCore::Allocator {
|
||||
inline void *malloc(size_t size) { return ::malloc(size); }
|
||||
inline void *calloc(size_t n, size_t size) { return ::calloc(n, size); }
|
||||
inline void *memalign(size_t align, size_t s) { return ::memalign(align, s); }
|
||||
inline void *valloc(size_t size) { return ::valloc(size); }
|
||||
inline void *valloc(size_t size)
|
||||
{
|
||||
#ifdef __ANDROID__
|
||||
return ::aligned_alloc(4096, size);
|
||||
#else
|
||||
return ::valloc(size);
|
||||
#endif
|
||||
}
|
||||
inline int posix_memalign(void** r, size_t a, size_t s) { return ::posix_memalign(r, a, s); }
|
||||
inline void *realloc(void* ptr, size_t size) { return ::realloc(ptr, size); }
|
||||
inline void free(void* ptr) { return ::free(ptr); }
|
||||
|
||||
@@ -27,6 +27,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
constexpr uint32_t LDAR_INST = 0x08'DF'FC'00;
|
||||
constexpr uint32_t LDAPR_INST = 0x38'BF'C0'00;
|
||||
constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
@@ -11,6 +11,91 @@
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
using ForkableUniqueMutex = std::mutex;
|
||||
using ForkableSharedMutex = std::shared_mutex;
|
||||
#endif
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedDeferredSignalWithMutexBase final {
|
||||
@@ -66,6 +151,20 @@ namespace FEXCore {
|
||||
using ScopedDeferredSignalWithSharedLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithUniqueLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedDeferredSignalWithForkableMutex = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedDeferredSignalWithForkableSharedLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithForkableUniqueLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedPotentialDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
@@ -131,4 +230,18 @@ namespace FEXCore {
|
||||
using ScopedPotentialDeferredSignalWithMutex = ScopedPotentialDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithSharedLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedPotentialDeferredSignalWithForkableMutex = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithForkableSharedLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithForkableUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
+44
-42
@@ -301,6 +301,33 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Add/subtract immediate") {
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r28, 4095, true), "cmp x28, #0xfff000 (16773120)");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r28, 16773120), "cmp x28, #0xfff000 (16773120)");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Min/max immediate") {
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, 1), "smax w29, w28, #1");
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, 127), "smax w29, w28, #127");
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, -128), "smax w29, w28, #-128");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, 1), "smax x29, x28, #1");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, 127), "smax x29, x28, #127");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, -128), "smax x29, x28, #-128");
|
||||
|
||||
TEST_SINGLE(umax(Size::i32Bit, Reg::r29, Reg::r28, 0), "umax w29, w28, #0");
|
||||
TEST_SINGLE(umax(Size::i32Bit, Reg::r29, Reg::r28, 255), "umax w29, w28, #255");
|
||||
TEST_SINGLE(umax(Size::i64Bit, Reg::r29, Reg::r28, 0), "umax x29, x28, #0");
|
||||
TEST_SINGLE(umax(Size::i64Bit, Reg::r29, Reg::r28, 255), "umax x29, x28, #255");
|
||||
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, 1), "smin w29, w28, #1");
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, 127), "smin w29, w28, #127");
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, -128), "smin w29, w28, #-128");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, 1), "smin x29, x28, #1");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, 127), "smin x29, x28, #127");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, -128), "smin x29, x28, #-128");
|
||||
|
||||
TEST_SINGLE(umin(Size::i32Bit, Reg::r29, Reg::r28, 0), "umin w29, w28, #0");
|
||||
TEST_SINGLE(umin(Size::i32Bit, Reg::r29, Reg::r28, 255), "umin w29, w28, #255");
|
||||
TEST_SINGLE(umin(Size::i64Bit, Reg::r29, Reg::r28, 0), "umin x29, x28, #0");
|
||||
TEST_SINGLE(umin(Size::i64Bit, Reg::r29, Reg::r28, 255), "umin x29, x28, #255");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Logical immediate") {
|
||||
TEST_SINGLE(and_(Size::i32Bit, Reg::r29, Reg::r28, 1), "and w29, w28, #0x1");
|
||||
TEST_SINGLE(and_(Size::i32Bit, Reg::r29, Reg::r28, -2), "and w29, w28, #0xfffffffe");
|
||||
@@ -428,6 +455,10 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Data processing - 2 source") {
|
||||
TEST_SINGLE(crc32cb(WReg::w29, WReg::w28, WReg::w27), "crc32cb w29, w28, w27");
|
||||
TEST_SINGLE(crc32ch(WReg::w29, WReg::w28, WReg::w27), "crc32ch w29, w28, w27");
|
||||
TEST_SINGLE(crc32cw(WReg::w29, WReg::w28, WReg::w27), "crc32cw w29, w28, w27");
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "smax w29, w28, w27");
|
||||
TEST_SINGLE(umax(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "umax w29, w28, w27");
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "smin w29, w28, w27");
|
||||
TEST_SINGLE(umin(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "umin w29, w28, w27");
|
||||
|
||||
TEST_SINGLE(udiv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "udiv x29, x28, x27");
|
||||
TEST_SINGLE(sdiv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "sdiv x29, x28, x27");
|
||||
@@ -435,6 +466,10 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Data processing - 2 source") {
|
||||
TEST_SINGLE(lsrv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "lsr x29, x28, x27");
|
||||
TEST_SINGLE(asrv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "asr x29, x28, x27");
|
||||
TEST_SINGLE(rorv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "ror x29, x28, x27");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "smax x29, x28, x27");
|
||||
TEST_SINGLE(umax(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "umax x29, x28, x27");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "smin x29, x28, x27");
|
||||
TEST_SINGLE(umin(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "umin x29, x28, x27");
|
||||
|
||||
if (false) {
|
||||
// vixl doesn't support this instruction.
|
||||
@@ -471,6 +506,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Data processing - 1 source") {
|
||||
TEST_SINGLE(rev(XReg::x29, XReg::x28), "rev x29, x28");
|
||||
TEST_SINGLE(rev(Size::i32Bit, Reg::r29, Reg::r28), "rev w29, w28");
|
||||
TEST_SINGLE(rev(Size::i64Bit, Reg::r29, Reg::r28), "rev x29, x28");
|
||||
|
||||
TEST_SINGLE(ctz(Size::i32Bit, Reg::r29, Reg::r28), "ctz w29, w28");
|
||||
TEST_SINGLE(ctz(Size::i64Bit, Reg::r29, Reg::r28), "ctz x29, x28");
|
||||
|
||||
TEST_SINGLE(cnt(Size::i32Bit, Reg::r29, Reg::r28), "cnt w29, w28");
|
||||
TEST_SINGLE(cnt(Size::i64Bit, Reg::r29, Reg::r28), "cnt x29, x28");
|
||||
|
||||
TEST_SINGLE(abs(Size::i32Bit, Reg::r29, Reg::r28), "abs w29, w28");
|
||||
TEST_SINGLE(abs(Size::i64Bit, Reg::r29, Reg::r28), "abs x29, x28");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PAUTH") {
|
||||
// TODO: Implement in the emitter.
|
||||
@@ -782,24 +826,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "add x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "add w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "add x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "add w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "add x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "add w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "add x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "add w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "add x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "add w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -814,24 +852,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "adds x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "adds w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "adds x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "adds w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "adds x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "adds w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "adds x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "adds w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "adds x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "adds w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -849,24 +881,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "sub x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "sub w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "sub x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "sub w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "sub x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "sub w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "sub x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "sub w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "sub x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "sub w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -881,24 +907,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "subs x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "subs w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "subs x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "subs w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "subs x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "subs w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "subs x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "subs w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "subs x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "subs w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -913,24 +933,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSL, 63), "neg x30, x29, lsl #63");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 31), "neg w30, w29, lsl #31");
|
||||
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "neg x30, x29, lsr #1");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "neg w30, w29, lsr #1");
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 63), "neg x30, x29, lsr #63");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 31), "neg w30, w29, lsr #31");
|
||||
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "neg x30, x29, asr #1");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "neg w30, w29, asr #1");
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 63), "neg x30, x29, asr #63");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 31), "neg w30, w29, asr #31");
|
||||
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -945,24 +959,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSL, 63), "cmp x30, x29, lsl #63");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 31), "cmp w30, w29, lsl #31");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "cmp x30, x29, lsr #1");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "cmp w30, w29, lsr #1");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 63), "cmp x30, x29, lsr #63");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 31), "cmp w30, w29, lsr #31");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "cmp x30, x29, asr #1");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "cmp w30, w29, asr #1");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 63), "cmp x30, x29, asr #63");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 31), "cmp w30, w29, asr #31");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -977,24 +985,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSL, 63), "negs x30, x29, lsl #63");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 31), "negs w30, w29, lsl #31");
|
||||
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "negs x30, x29, lsr #1");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "negs w30, w29, lsr #1");
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 63), "negs x30, x29, lsr #63");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 31), "negs w30, w29, lsr #31");
|
||||
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "negs x30, x29, asr #1");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "negs w30, w29, asr #1");
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 63), "negs x30, x29, asr #63");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 31), "negs w30, w29, asr #31");
|
||||
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
|
||||
+1
-1
@@ -2110,7 +2110,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE FFR write from predicate")
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE FFR initialise") {
|
||||
TEST_SINGLE(setffr(), "setffr ");
|
||||
TEST_SINGLE(setffr(), "setffr");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE Integer Multiply-Add - Unpredicated") {
|
||||
|
||||
+8
-8
@@ -31,16 +31,16 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System Instruction") {
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGSW, Reg::r30), "sys #0, C7, C14, #4, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGDSW, Reg::r30), "sys #0, C7, C14, #6, x30");
|
||||
|
||||
TEST_SINGLE(dc(DataCacheOperation::GVA, Reg::r30), "sys #3, C7, C4, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::GZVA, Reg::r30), "sys #3, C7, C4, #4, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAC, Reg::r30), "sys #3, C7, C10, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAC, Reg::r30), "sys #3, C7, C10, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAP, Reg::r30), "sys #3, C7, C12, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAP, Reg::r30), "sys #3, C7, C12, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::GVA, Reg::r30), "dc gva, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::GZVA, Reg::r30), "dc gzva, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAC, Reg::r30), "dc cgvac, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAC, Reg::r30), "dc cgdvac, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAP, Reg::r30), "dc cgvap, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAP, Reg::r30), "dc cgdvap, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVADP, Reg::r30), "sys #3, C7, C13, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVADP, Reg::r30), "sys #3, C7, C13, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGVAC, Reg::r30), "sys #3, C7, C14, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGDVAC, Reg::r30), "sys #3, C7, C14, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGVAC, Reg::r30), "dc cigvac, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGDVAC, Reg::r30), "dc cigdvac, x30");
|
||||
|
||||
TEST_SINGLE(dc(DataCacheOperation::CVAP, Reg::r30), "dc cvap, x30");
|
||||
|
||||
|
||||
Vendored
+1
-1
Submodule External/fmt updated: c4ee726532...a0b8a92e3d.
Vendored
+1
-1
Submodule External/jemalloc updated: 19196715f1...16f8061955.
Vendored
+1
-1
Submodule External/jemalloc_glibc updated: 79e02853ed...888181c5f7.
Vendored
+1
-1
Submodule External/vixl updated: 1027d946a5...96f22fe65d.
@@ -265,7 +265,7 @@ namespace FHU::Filesystem {
|
||||
size_t DataSize = (sizeof(std::string_view) + sizeof(void*) * 2) * (SeparatorCount + 2);
|
||||
void *Data = alloca(DataSize);
|
||||
fextl::pmr::fixed_size_monotonic_buffer_resource mbr(Data, DataSize);
|
||||
std::pmr::polymorphic_allocator pa {&mbr};
|
||||
std::pmr::polymorphic_allocator<std::byte> pa {&mbr};
|
||||
std::pmr::list<std::string_view> Parts{pa};
|
||||
|
||||
size_t CurrentOffset{};
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
@@ -106,4 +106,17 @@ namespace FHU {
|
||||
using ScopedSignalMaskWithMutex = ScopedSignalMaskWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedSignalMaskWithSharedLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithUniqueLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
using ScopedSignalMaskWithForkableMutex = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedSignalMaskWithForkableSharedLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithForkableUniqueLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -9,6 +9,7 @@ from threading import Thread
|
||||
import subprocess
|
||||
import time
|
||||
import multiprocessing
|
||||
from shutil import which
|
||||
|
||||
if sys.version_info[0] < 3:
|
||||
raise Exception("Python 3 or a more recent version is required.")
|
||||
@@ -41,6 +42,10 @@ def Threaded_Manager(Runner, ID, File):
|
||||
ServerArgs = ["catchsegv", Runner, "-c", "vm", "-n", "1", "-I", "R" + str(ID), File]
|
||||
ClientArgs = ["catchsegv", Runner, "-c", "vm", "-n", "1", "-I", "R" + str(ID), "-C"]
|
||||
|
||||
if which("catchsegv") is None:
|
||||
ServerArgs.pop(0)
|
||||
ClientArgs.pop(0)
|
||||
|
||||
ServerThread = Thread(target = Threaded_Runner, args = (ServerArgs, ID, 0))
|
||||
ClientThread = Thread(target = Threaded_Runner, args = (ClientArgs, ID, 1))
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@ import sys
|
||||
import subprocess
|
||||
import os.path
|
||||
from os import path
|
||||
from shutil import which
|
||||
|
||||
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
|
||||
|
||||
@@ -46,6 +47,9 @@ if path.exists(disabled_tests_runner_file):
|
||||
disabled_tests[line.strip()] = 1
|
||||
|
||||
RunnerArgs = ["catchsegv", runner]
|
||||
|
||||
if which("catchsegv") is None:
|
||||
RunnerArgs.pop(0)
|
||||
# Add the rest of the arguments
|
||||
for i in range(len(sys.argv) - args_start_index):
|
||||
RunnerArgs.append(sys.argv[args_start_index + i])
|
||||
|
||||
@@ -14,6 +14,6 @@ if (NOT MINGW_BUILD)
|
||||
endif()
|
||||
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse json-maker FEXHeaderUtils)
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse tiny-json json-maker FEXHeaderUtils)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/External/cpp-optparse/)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_BINARY_DIR}/generated)
|
||||
+232
-7
@@ -5,6 +5,7 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
#include <FEXHeaderUtils/SymlinkChecks.h>
|
||||
|
||||
@@ -14,8 +15,74 @@
|
||||
#include <pwd.h>
|
||||
#include <utility>
|
||||
#include <json-maker.h>
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEX::Config {
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static const fextl::map<FEXCore::Config::ConfigOption, fextl::string> ConfigToNameLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {FEXCore::Config::ConfigOption::CONFIG_##enum, #json},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
@@ -45,6 +112,164 @@ namespace FEX::Config {
|
||||
}
|
||||
}
|
||||
|
||||
// Application loaders
|
||||
class OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit OptionMapper(FEXCore::Config::LayerType Layer);
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char *ConfigName, const char *ConfigString);
|
||||
};
|
||||
|
||||
class MainLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return fextl::make_unique<MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<EnvLoader>(_envp);
|
||||
}
|
||||
|
||||
fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInterp, const std::string_view ProgramFDFromEnv) {
|
||||
// If executed with a FEX FD then the Program argument might be empty.
|
||||
// In this case we need to scan the FD node to recover the application binary that exists on disk.
|
||||
@@ -120,8 +345,8 @@ namespace FEX::Config {
|
||||
const std::string_view ProgramFDFromEnv) {
|
||||
FEX::Config::InitializeConfigs();
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(CreateMainLayer());
|
||||
|
||||
if (NoFEXArguments) {
|
||||
FEX::ArgLoader::LoadWithoutArguments(argc, argv);
|
||||
@@ -130,7 +355,7 @@ namespace FEX::Config {
|
||||
FEXCore::Config::AddLayer(fextl::make_unique<FEX::ArgLoader::ArgLoader>(argc, argv));
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
auto Args = FEX::ArgLoader::Get();
|
||||
@@ -183,16 +408,16 @@ namespace FEX::Config {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
|
||||
auto SteamID = getenv("SteamAppId");
|
||||
if (SteamID) {
|
||||
// If a SteamID exists then let's search for Steam application configs as well.
|
||||
// We want to key off both the SteamAppId number /and/ the executable since we may not want to thunk all binaries.
|
||||
fextl::string SteamAppName = fextl::fmt::format("Steam_{}_{}", SteamID, ProgramName);
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
}
|
||||
|
||||
return ApplicationNames{std::move(Program), std::move(ProgramName)};
|
||||
|
||||
@@ -55,4 +55,40 @@ namespace FEX::Config {
|
||||
fextl::string GetConfigFileLocation(bool Global);
|
||||
|
||||
void InitializeConfigs();
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
|
||||
/**
|
||||
* @brief Loads the main application config
|
||||
*
|
||||
* @param File Optional override to load a specific config file in to the main layer
|
||||
* Shouldn't be commonly used
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
*
|
||||
* @param Filename Application filename component
|
||||
* @param Global Load the global configuration or user accessible file
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
*
|
||||
* @param _envp[] The environment array from main
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]);
|
||||
}
|
||||
@@ -153,7 +153,7 @@ namespace FEXServerClient {
|
||||
auto ServerSocketName = GetServerSocketName();
|
||||
|
||||
// Create the initial unix socket
|
||||
int SocketFD = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
int SocketFD = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (SocketFD == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
return -1;
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
if (NOT MINGW_BUILD)
|
||||
if (NOT TERMUX_BUILD)
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
if (BUILD_FEXCONFIG)
|
||||
add_subdirectory(FEXConfig/)
|
||||
endif()
|
||||
|
||||
if (NOT TERMUX_BUILD)
|
||||
# Disable FEXRootFSFetcher on Termux, it doesn't even work there
|
||||
add_subdirectory(FEXRootFSFetcher/)
|
||||
endif()
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
|
||||
#include "ConfigDefines.h"
|
||||
#include "Common/ArgumentLoader.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FEXServerClient.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -23,10 +24,10 @@ $end_info$
|
||||
|
||||
int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateMainLayer());
|
||||
FEX::ArgLoader::LoadWithoutArguments(argc, argv);
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
// Reload the meta layer
|
||||
|
||||
@@ -99,7 +99,7 @@ namespace {
|
||||
}
|
||||
ConfigOpen = true;
|
||||
ConfigFilename = Filename;
|
||||
LoadedConfig = FEXCore::Config::CreateMainLayer(&Filename);
|
||||
LoadedConfig = FEX::Config::CreateMainLayer(&Filename);
|
||||
LoadedConfig->Load();
|
||||
|
||||
// Load default options and only overwrite only if the option didn't exist
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#include "ConfigDefines.h"
|
||||
#include "Common/cpp-optparse/OptionParser.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FEXServerClient.h"
|
||||
#include "FEXHeaderUtils/Filesystem.h"
|
||||
#include "git_version.h"
|
||||
@@ -13,10 +14,10 @@
|
||||
|
||||
int main(int argc, char **argv, char **envp) {
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateMainLayer());
|
||||
// No FEX arguments passed through command line
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
// Load the arguments
|
||||
@@ -43,16 +44,16 @@ int main(int argc, char **argv, char **envp) {
|
||||
if (Options.is_set_by_user("app")) {
|
||||
// Load the application config if one was provided
|
||||
const auto ProgramName = FHU::Filesystem::GetFilename(Options["app"]);
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
|
||||
auto SteamID = getenv("SteamAppId");
|
||||
if (SteamID) {
|
||||
// If a SteamID exists then let's search for Steam application configs as well.
|
||||
// We want to key off both the SteamAppId number /and/ the executable since we may not want to thunk all binaries.
|
||||
const auto SteamAppName = fextl::fmt::format("Steam_{}_{}", SteamID, ProgramName);
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -171,7 +171,7 @@ int main(int argc, char **argv, char **const envp)
|
||||
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(fextl::make_unique<FEX::ArgLoader::ArgLoader>(argc, argv));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
// Ensure the IRLoader runs in 64-bit mode.
|
||||
// This is to ensure that static register allocation in the JIT
|
||||
|
||||
@@ -31,6 +31,7 @@ $end_info$
|
||||
#include <stdio.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/statfs.h>
|
||||
#include <sys/xattr.h>
|
||||
#include <syscall.h>
|
||||
#include <system_error>
|
||||
#include <unistd.h>
|
||||
@@ -592,7 +593,7 @@ uint64_t FileManager::FAccessat(int dirfd, const char *pathname, int mode) {
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dirfd, SelfPath, true, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(faccessat2), Path.first, Path.second, mode, 0);
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(faccessat), Path.first, Path.second, mode);
|
||||
if (Result != -1)
|
||||
return Result;
|
||||
}
|
||||
@@ -842,4 +843,124 @@ uint64_t FileManager::NewFSStatAt64(int dirfd, const char *pathname, struct stat
|
||||
return ::fstatat64(dirfd, SelfPath, buf, flag);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Setxattr(const char *path, const char *name, const void *value, size_t size, int flags) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, true);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::setxattr(Path.c_str(), name, value, size, flags);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::setxattr(SelfPath, name, value, size, flags);
|
||||
}
|
||||
|
||||
uint64_t FileManager::LSetxattr(const char *path, const char *name, const void *value, size_t size, int flags) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, false);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::lsetxattr(Path.c_str(), name, value, size, flags);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::lsetxattr(SelfPath, name, value, size, flags);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Getxattr(const char *path, const char *name, void *value, size_t size) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, true);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::getxattr(Path.c_str(), name, value, size);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::getxattr(SelfPath, name, value, size);
|
||||
}
|
||||
|
||||
uint64_t FileManager::LGetxattr(const char *path, const char *name, void *value, size_t size) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, false);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::lgetxattr(Path.c_str(), name, value, size);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::lgetxattr(SelfPath, name, value, size);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Listxattr(const char *path, char *list, size_t size) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, true);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::listxattr(Path.c_str(), list, size);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::listxattr(SelfPath, list, size);
|
||||
}
|
||||
|
||||
uint64_t FileManager::LListxattr(const char *path, char *list, size_t size) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, false);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::llistxattr(Path.c_str(), list, size);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::llistxattr(SelfPath, list, size);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Removexattr(const char *path, const char *name) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, true);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::removexattr(Path.c_str(), name);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::removexattr(SelfPath, name);
|
||||
}
|
||||
|
||||
uint64_t FileManager::LRemovexattr(const char *path, const char *name) {
|
||||
auto NewPath = GetSelf(path);
|
||||
const char *SelfPath = NewPath ? NewPath->c_str() : nullptr;
|
||||
|
||||
auto Path = GetEmulatedPath(SelfPath, false);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::lremovexattr(Path.c_str(), name);
|
||||
if (Result != -1 || errno != ENOENT) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
return ::lremovexattr(SelfPath, name);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -66,7 +66,14 @@ public:
|
||||
uint64_t Mknod(const char *pathname, mode_t mode, dev_t dev);
|
||||
uint64_t NewFSStatAt(int dirfd, const char *pathname, struct stat *buf, int flag);
|
||||
uint64_t NewFSStatAt64(int dirfd, const char *pathname, struct stat64 *buf, int flag);
|
||||
|
||||
uint64_t Setxattr(const char *path, const char *name, const void *value, size_t size, int flags);
|
||||
uint64_t LSetxattr(const char *path, const char *name, const void *value, size_t size, int flags);
|
||||
uint64_t Getxattr(const char *path, const char *name, void *value, size_t size);
|
||||
uint64_t LGetxattr(const char *path, const char *name, void *value, size_t size);
|
||||
uint64_t Listxattr(const char *path, char *list, size_t size);
|
||||
uint64_t LListxattr(const char *path, char *list, size_t size);
|
||||
uint64_t Removexattr(const char *path, const char *name);
|
||||
uint64_t LRemovexattr(const char *path, const char *name);
|
||||
// vfs
|
||||
uint64_t Statfs(const char *path, void *buf);
|
||||
|
||||
|
||||
@@ -41,6 +41,7 @@ $end_info$
|
||||
|
||||
#include <algorithm>
|
||||
#include <alloca.h>
|
||||
#include <charconv>
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
#include <regex>
|
||||
@@ -579,12 +580,33 @@ uint64_t CloneHandler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clone3_args
|
||||
if (!AnyFlagsSet(flags, CLONE_THREAD)) {
|
||||
// Has an unsupported flag
|
||||
// Fall to a handler that can handle this case
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
args->SignalMask = ~0ULL;
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &args->SignalMask, &args->SignalMask, sizeof(args->SignalMask));
|
||||
Thread->CTX->LockBeforeFork(Frame->Thread);
|
||||
|
||||
FEX::HLE::_SyscallHandler->LockBeforeFork();
|
||||
|
||||
uint64_t Result{};
|
||||
if (args->Type == TYPE_CLONE2) {
|
||||
return Clone2Handler(Frame, args);
|
||||
Result = Clone2Handler(Frame, args);
|
||||
}
|
||||
else {
|
||||
return Clone3Handler(Frame, args);
|
||||
Result = Clone3Handler(Frame, args);
|
||||
}
|
||||
|
||||
if (Result != 0) {
|
||||
// Parent
|
||||
// Unlock the mutexes on both sides of the fork
|
||||
FEX::HLE::_SyscallHandler->UnlockAfterFork(false);
|
||||
|
||||
// Clear all the other threads that are being tracked
|
||||
Thread->CTX->UnlockAfterFork(Frame->Thread, false);
|
||||
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &args->SignalMask, nullptr, sizeof(args->SignalMask));
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::IFmt("Unsupported flag with CLONE_THREAD. This breaks TLS, falling down classic thread path");
|
||||
@@ -606,12 +628,6 @@ uint64_t CloneHandler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clone3_args
|
||||
}
|
||||
|
||||
if (!(flags & CLONE_THREAD)) {
|
||||
if (flags & CLONE_VFORK) {
|
||||
PrintFlags(flags);
|
||||
flags &= ~CLONE_VM;
|
||||
LogMan::Msg::DFmt("clone: WARNING: CLONE_VFORK w/o CLONE_THREAD");
|
||||
}
|
||||
|
||||
// CLONE_PARENT is ignored (Implied by CLONE_THREAD)
|
||||
return FEX::HLE::ForkGuest(Thread, Frame, flags,
|
||||
reinterpret_cast<void*>(args->args.stack),
|
||||
@@ -639,6 +655,7 @@ uint64_t CloneHandler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clone3_args
|
||||
// Normally a thread cleans itself up on exit. But because we need to join, we are now responsible
|
||||
Thread->CTX->DestroyThread(NewThread);
|
||||
}
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
}
|
||||
};
|
||||
@@ -741,16 +758,16 @@ uint32_t SyscallHandler::CalculateHostKernelVersion() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
int32_t Major{};
|
||||
int32_t Minor{};
|
||||
int32_t Patch{};
|
||||
char Tmp{};
|
||||
fextl::istringstream ss{buf.release};
|
||||
ss >> Major;
|
||||
ss.read(&Tmp, 1);
|
||||
ss >> Minor;
|
||||
ss.read(&Tmp, 1);
|
||||
ss >> Patch;
|
||||
uint32_t Major{};
|
||||
uint32_t Minor{};
|
||||
uint32_t Patch{};
|
||||
|
||||
// Parse kernel version in the form of `<Major>.<Minor>.<Patch>[Optional Data]`
|
||||
const auto End = buf.release + sizeof(buf.release);
|
||||
auto Results = std::from_chars(buf.release, End, Major, 10);
|
||||
Results = std::from_chars(Results.ptr + 1, End, Minor, 10);
|
||||
Results = std::from_chars(Results.ptr + 1, End, Patch, 10);
|
||||
|
||||
return (Major << 24) | (Minor << 16) | Patch;
|
||||
}
|
||||
|
||||
@@ -813,17 +830,16 @@ uint64_t UnimplementedSyscallSafe(FEXCore::Core::CpuStateFrame *Frame, uint64_t
|
||||
}
|
||||
|
||||
void SyscallHandler::LockBeforeFork() {
|
||||
// XXX shared_mutex has issues with locking and forks
|
||||
// VMATracking.Mutex.lock();
|
||||
|
||||
// Add other mutexes here
|
||||
VMATracking.Mutex.lock();
|
||||
}
|
||||
|
||||
void SyscallHandler::UnlockAfterFork() {
|
||||
// Add other mutexes here
|
||||
|
||||
// XXX shared_mutex has issues with locking and forks
|
||||
// VMATracking.Mutex.unlock();
|
||||
void SyscallHandler::UnlockAfterFork(bool Child) {
|
||||
if (Child) {
|
||||
VMATracking.Mutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
VMATracking.Mutex.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
static bool isHEX(char c) {
|
||||
|
||||
@@ -15,6 +15,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -220,7 +221,7 @@ public:
|
||||
|
||||
///// FORK tracking /////
|
||||
void LockBeforeFork();
|
||||
void UnlockAfterFork();
|
||||
void UnlockAfterFork(bool Child);
|
||||
|
||||
SourcecodeResolver *GetSourcecodeResolver() override { return this; }
|
||||
|
||||
@@ -327,7 +328,7 @@ private:
|
||||
struct VMATracking {
|
||||
using VMAEntry = SyscallHandler::VMAEntry;
|
||||
// Held while reading/writing this struct
|
||||
std::shared_mutex Mutex;
|
||||
FEXCore::ForkableSharedMutex Mutex;
|
||||
|
||||
// Memory ranges indexed by page aligned starting address
|
||||
fextl::map<uint64_t, VMAEntry> VMAs;
|
||||
@@ -430,6 +431,7 @@ enum TypeOfClone {
|
||||
|
||||
struct clone3_args {
|
||||
TypeOfClone Type;
|
||||
uint64_t SignalMask;
|
||||
kernel_clone3_args args;
|
||||
};
|
||||
|
||||
|
||||
@@ -181,15 +181,15 @@ namespace FEX::HLE {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(setxattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(setxattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, const char *name, const void *value, size_t size, int flags) -> uint64_t {
|
||||
uint64_t Result = ::setxattr(path, name, value, size, flags);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.Setxattr(path, name, value, size, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(lsetxattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(lsetxattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, const char *name, const void *value, size_t size, int flags) -> uint64_t {
|
||||
uint64_t Result = ::lsetxattr(path, name, value, size, flags);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.LSetxattr(path, name, value, size, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
@@ -199,15 +199,15 @@ namespace FEX::HLE {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(getxattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(getxattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, const char *name, void *value, size_t size) -> uint64_t {
|
||||
uint64_t Result = ::getxattr(path, name, value, size);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.Getxattr(path, name, value, size);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(lgetxattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(lgetxattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, const char *name, void *value, size_t size) -> uint64_t {
|
||||
uint64_t Result = ::lgetxattr(path, name, value, size);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.LGetxattr(path, name, value, size);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
@@ -217,15 +217,15 @@ namespace FEX::HLE {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(listxattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(listxattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, char *list, size_t size) -> uint64_t {
|
||||
uint64_t Result = ::listxattr(path, list, size);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.Listxattr(path, list, size);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(llistxattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(llistxattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, char *list, size_t size) -> uint64_t {
|
||||
uint64_t Result = ::llistxattr(path, list, size);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.LListxattr(path, list, size);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
@@ -235,15 +235,15 @@ namespace FEX::HLE {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(removexattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(removexattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, const char *name) -> uint64_t {
|
||||
uint64_t Result = ::removexattr(path, name);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.Removexattr(path, name);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(lremovexattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
REGISTER_SYSCALL_IMPL(lremovexattr,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, const char *path, const char *name) -> uint64_t {
|
||||
uint64_t Result = ::lremovexattr(path, name);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.LRemovexattr(path, name);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -140,7 +140,13 @@ namespace FEX::HLE {
|
||||
// If we don't have CLONE_THREAD then we are effectively a fork
|
||||
// Clear all the other threads that are being tracked
|
||||
// Frame->Thread is /ONLY/ safe to access when CLONE_THREAD flag is not set
|
||||
CTX->CleanupAfterFork(Frame->Thread);
|
||||
// Unlock the mutexes on both sides of the fork
|
||||
FEX::HLE::_SyscallHandler->UnlockAfterFork(true);
|
||||
|
||||
// Clear all the other threads that are being tracked
|
||||
Thread->CTX->UnlockAfterFork(Frame->Thread, true);
|
||||
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &CloneArgs->SignalMask, nullptr, sizeof(CloneArgs->SignalMask));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RAX] = 0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RBX] = 0;
|
||||
@@ -187,6 +193,11 @@ namespace FEX::HLE {
|
||||
|
||||
uint64_t ForkGuest(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::CpuStateFrame *Frame, uint32_t flags, void *stack, size_t StackSize, pid_t *parent_tid, pid_t *child_tid, void *tls) {
|
||||
// Just before we fork, we lock all syscall mutexes so that both processes will end up with a locked mutex
|
||||
|
||||
uint64_t Mask{~0ULL};
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &Mask, sizeof(Mask));
|
||||
Thread->CTX->LockBeforeFork(Frame->Thread);
|
||||
|
||||
FEX::HLE::_SyscallHandler->LockBeforeFork();
|
||||
|
||||
const bool IsVFork = flags & CLONE_VFORK;
|
||||
@@ -215,11 +226,17 @@ namespace FEX::HLE {
|
||||
else {
|
||||
Result = fork();
|
||||
}
|
||||
const bool IsChild = Result == 0;
|
||||
|
||||
// Unlock the mutexes on both sides of the fork
|
||||
FEX::HLE::_SyscallHandler->UnlockAfterFork();
|
||||
if (IsChild) {
|
||||
// Unlock the mutexes on both sides of the fork
|
||||
FEX::HLE::_SyscallHandler->UnlockAfterFork(IsChild);
|
||||
|
||||
// Clear all the other threads that are being tracked
|
||||
Thread->CTX->UnlockAfterFork(Frame->Thread, IsChild);
|
||||
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, nullptr, sizeof(Mask));
|
||||
|
||||
if (Result == 0) {
|
||||
// Child
|
||||
// update the internal TID
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
@@ -227,9 +244,6 @@ namespace FEX::HLE {
|
||||
FEX::HLE::_SyscallHandler->FM.UpdatePID(Thread->ThreadManager.PID);
|
||||
Thread->ThreadManager.clear_child_tid = nullptr;
|
||||
|
||||
// Clear all the other threads that are being tracked
|
||||
Thread->CTX->CleanupAfterFork(Frame->Thread);
|
||||
|
||||
// only a single thread running so no need to remove anything from the thread array
|
||||
|
||||
// Handle child setup now
|
||||
@@ -275,6 +289,14 @@ namespace FEX::HLE {
|
||||
}
|
||||
}
|
||||
|
||||
// Unlock the mutexes on both sides of the fork
|
||||
FEX::HLE::_SyscallHandler->UnlockAfterFork(IsChild);
|
||||
|
||||
// Clear all the other threads that are being tracked
|
||||
Thread->CTX->UnlockAfterFork(Frame->Thread, IsChild);
|
||||
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, nullptr, sizeof(Mask));
|
||||
|
||||
// VFork needs the parent to wait for the child to exit.
|
||||
if (IsVFork) {
|
||||
// Wait for the read end of the pipe to close.
|
||||
|
||||
@@ -54,7 +54,7 @@ bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState *Thread,
|
||||
|
||||
{
|
||||
// Can't use the deferred signal lock in the SIGSEGV handler.
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(_SyscallHandler->VMATracking.Mutex);
|
||||
FHU::ScopedSignalMaskWithForkableSharedLock lk(_SyscallHandler->VMATracking.Mutex);
|
||||
|
||||
auto VMATracking = &_SyscallHandler->VMATracking;
|
||||
|
||||
@@ -111,7 +111,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::ScopedDeferredSignalWithSharedLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedDeferredSignalWithForkableSharedLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
// Find the first mapping at or after the range ends, or ::end().
|
||||
// Top points to the address after the end of the range
|
||||
@@ -166,7 +166,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
|
||||
|
||||
// Used for AOT
|
||||
FEXCore::HLE::AOTIRCacheEntryLookupResult SyscallHandler::LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestAddr) {
|
||||
FEXCore::ScopedDeferredSignalWithSharedLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedDeferredSignalWithForkableSharedLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
// Get the first mapping after GuestAddr, or end
|
||||
// GuestAddr is inclusive
|
||||
@@ -194,7 +194,7 @@ void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState *Thread, uintp
|
||||
// NOTE: Frontend calls this with a nullptr Thread during initialization, but
|
||||
// providing this code with a valid Thread object earlier would allow
|
||||
// us to be more optimal by using ScopedDeferredSignalWithUniqueLock instead
|
||||
FEXCore::ScopedPotentialDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
static uint64_t AnonSharedId = 1;
|
||||
|
||||
@@ -233,6 +233,7 @@ void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState *Thread, uintp
|
||||
}
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
// VMATracking.Mutex can't be held while executing this, otherwise it hangs if the JIT is in the process of looking up code in the AOT JIT.
|
||||
CTX->InvalidateGuestCodeRange(Thread, (uintptr_t)Base, Size);
|
||||
}
|
||||
}
|
||||
@@ -244,7 +245,7 @@ void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
// Frontend calls this with nullptr Thread during initialization.
|
||||
// This is why `ScopedPotentialDeferredSignalWithUniqueLock` is used here.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
FEXCore::ScopedPotentialDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
VMATracking.ClearUnsafe(CTX, Base, Size);
|
||||
}
|
||||
@@ -258,7 +259,7 @@ void SyscallHandler::TrackMprotect(FEXCore::Core::InternalThreadState *Thread, u
|
||||
Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
VMATracking.ChangeUnsafe(Base, Size, VMAProt::fromProt(Prot));
|
||||
}
|
||||
@@ -273,7 +274,7 @@ void SyscallHandler::TrackMremap(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
NewSize = FEXCore::AlignUp(NewSize, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
const auto OldVMA = VMATracking.LookupVMAUnsafe(OldAddress);
|
||||
|
||||
@@ -331,7 +332,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState *Thread, int
|
||||
uint64_t Length = stat.shm_segsz;
|
||||
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
// TODO
|
||||
MRID mrid{SpecialDev::SHM, static_cast<uint64_t>(shmid)};
|
||||
@@ -353,7 +354,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState *Thread, int
|
||||
void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState *Thread, uintptr_t Base) {
|
||||
uintptr_t Length = 0;
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
Length = VMATracking.ClearShmUnsafe(CTX, Base);
|
||||
}
|
||||
@@ -367,8 +368,7 @@ void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState *Thread, uint
|
||||
void SyscallHandler::TrackMadvise(FEXCore::Core::InternalThreadState *Thread, uintptr_t Base, uintptr_t Size, int advice) {
|
||||
Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE);
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
// TODO
|
||||
}
|
||||
}
|
||||
|
||||
@@ -167,6 +167,11 @@ namespace FEX::HLE::x32 {
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(set_robust_list, [](FEXCore::Core::CpuStateFrame *Frame, struct robust_list_head *head, size_t len) -> uint64_t {
|
||||
if (len != 12) {
|
||||
// Return invalid if the passed in length doesn't match what's expected.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
auto Thread = Frame->Thread;
|
||||
// Retain the robust list head but don't give it to the kernel
|
||||
// The kernel would break if it tried parsing a 32bit robust list from a 64bit process
|
||||
@@ -178,7 +183,7 @@ namespace FEX::HLE::x32 {
|
||||
auto Thread = Frame->Thread;
|
||||
// Give the robust list back to the application
|
||||
// Steam specifically checks to make sure the robust list is set
|
||||
*(uint32_t**)head = (uint32_t*)Thread->ThreadManager.robust_list_head;
|
||||
*(uint32_t*)head = (uint32_t)Thread->ThreadManager.robust_list_head;
|
||||
*len_ptr = 12;
|
||||
return 0;
|
||||
});
|
||||
|
||||
@@ -207,7 +207,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
LogMan::Msg::InstallHandler(MsgHandler);
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(fextl::make_unique<FEX::ArgLoader::ArgLoader>(argc, argv));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
auto Args = FEX::ArgLoader::Get();
|
||||
|
||||
@@ -1076,7 +1076,7 @@ namespace {
|
||||
namespace ConfigSetter {
|
||||
void SetRootFSAsDefault(const fextl::string &RootFS) {
|
||||
fextl::string Filename = FEXCore::Config::GetConfigFileLocation();
|
||||
auto LoadedConfig = FEXCore::Config::CreateMainLayer(&Filename);
|
||||
auto LoadedConfig = FEX::Config::CreateMainLayer(&Filename);
|
||||
LoadedConfig->Load();
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_ROOTFS, RootFS);
|
||||
FEX::Config::SaveLayerToJSON(Filename, LoadedConfig.get());
|
||||
|
||||
@@ -6,15 +6,16 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include "Common/Config.h"
|
||||
#include "Common/ArgumentLoader.h"
|
||||
|
||||
#include <memory>
|
||||
|
||||
int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(fextl::make_unique<FEX::ArgLoader::ArgLoader>(argc, argv));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
# FEX-2306
|
||||
# FEX-2307
|
||||
|
||||
## External/FEXCore
|
||||
See [FEXCore/Readme.md](../External/FEXCore/Readme.md) for more details
|
||||
|
||||
@@ -38,7 +38,7 @@ foreach(ASM_SRC ${ASM_SOURCES})
|
||||
|
||||
add_custom_command(OUTPUT ${OUTPUT_NAME}
|
||||
DEPENDS "${TMP_FILE}"
|
||||
COMMAND "nasm" ARGS "${TMP_FILE}" "-o" "${OUTPUT_NAME}")
|
||||
COMMAND "nasm" ARGS "-i" "${CMAKE_SOURCE_DIR}/unittests/ASM/Includes/" "${TMP_FILE}" "-o" "${OUTPUT_NAME}")
|
||||
|
||||
add_custom_command(OUTPUT ${OUTPUT_CONFIG_NAME}
|
||||
DEPENDS "${ASM_SRC}"
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x00060F000F000D01", "0x0000000000070007"],
|
||||
"XMM1": ["0x3111313131311111", "0x0000000000313131"],
|
||||
"XMM0": ["0x00060F000F000D01", "0x0000001010070007"],
|
||||
"XMM1": ["0x3111313131311111", "0x0000001818313131"],
|
||||
"XMM2": ["0x005A0041007A0061", "0x55AACCBBFF220000"],
|
||||
"XMM3": ["0x006500200027003F", "0x00210065004F0065"]
|
||||
"XMM3": ["0x0065002000270000", "0x00210065004F0065"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -104,6 +104,16 @@ CompareAndStore 9, 0b00110101
|
||||
; Range unsigned word check (msb, negative masked)
|
||||
CompareAndStore 10, 0b01110101
|
||||
|
||||
; --- Edge case test (string begins with null character) ---
|
||||
movaps xmm2, [rel .data_null]
|
||||
movaps xmm3, [rel .data_null + 32]
|
||||
|
||||
; Range signed byte check (msb)
|
||||
CompareAndStore 11, 0b01000110
|
||||
|
||||
; Range signed byte check (lsb)
|
||||
CompareAndStore 12, 0b01000110
|
||||
|
||||
; Load all our stored indices and flags for result comparing
|
||||
movaps xmm0, [rel .indices]
|
||||
movaps xmm1, [rel .flags]
|
||||
@@ -133,6 +143,17 @@ dq 0x00210065004F0065 ; "eOen!"
|
||||
dq 0x8888888888888888
|
||||
dq 0x9999999999999999
|
||||
|
||||
.data_null:
|
||||
dq 0x005A0041007A0061 ; "azAZ"
|
||||
dq 0x55AACCBBFF220000
|
||||
dq 0xAAAAAAAAAAAAAAAA
|
||||
dq 0xBBBBBBBBBBBBBBBB
|
||||
|
||||
dq 0x0065002000270000 ; "\0' e"
|
||||
dq 0x00210065004F0065 ; "eOen!"
|
||||
dq 0x8888888888888888
|
||||
dq 0x9999999999999999
|
||||
|
||||
.indices:
|
||||
dq 0x0000000000000000
|
||||
dq 0x0000000000000000
|
||||
|
||||
@@ -0,0 +1,215 @@
|
||||
;
|
||||
; Various macros used to set up data for the XSAVE tests
|
||||
;
|
||||
; Define IS_AVX before including this file to enable the
|
||||
; use of AVX instructions to handle the upper lanes.
|
||||
;
|
||||
|
||||
%ifndef XSAVE_MACROS_INC
|
||||
%define XSAVE_MACROS_INC
|
||||
|
||||
; Initializes the MMX registers to various values using a label to a memory region
|
||||
%macro set_up_mmx_state 1
|
||||
movq mm0, [rel %1 + 32 * 0]
|
||||
movq mm1, [rel %1 + 32 * 1]
|
||||
movq mm2, [rel %1 + 32 * 2]
|
||||
movq mm3, [rel %1 + 32 * 3]
|
||||
movq mm4, [rel %1 + 32 * 4]
|
||||
movq mm5, [rel %1 + 32 * 5]
|
||||
movq mm6, [rel %1 + 32 * 6]
|
||||
movq mm7, [rel %1 + 32 * 7]
|
||||
%endmacro
|
||||
|
||||
; Sets up the XMM registers using a given label to a memory region.
|
||||
%macro set_up_xmm_state 1
|
||||
%macro move_to_xmm 2
|
||||
%ifdef IS_AVX
|
||||
vmovaps ymm%1, [rel %2 + 32 * %1]
|
||||
%else
|
||||
movaps xmm%1, [rel %2 + 32 * %1]
|
||||
%endif
|
||||
%endmacro
|
||||
|
||||
move_to_xmm 0, %1
|
||||
move_to_xmm 1, %1
|
||||
move_to_xmm 2, %1
|
||||
move_to_xmm 3, %1
|
||||
move_to_xmm 4, %1
|
||||
move_to_xmm 5, %1
|
||||
move_to_xmm 6, %1
|
||||
move_to_xmm 7, %1
|
||||
move_to_xmm 8, %1
|
||||
move_to_xmm 9, %1
|
||||
move_to_xmm 10, %1
|
||||
move_to_xmm 11, %1
|
||||
move_to_xmm 12, %1
|
||||
move_to_xmm 13, %1
|
||||
move_to_xmm 14, %1
|
||||
move_to_xmm 15, %1
|
||||
|
||||
%undef move_to_xmm
|
||||
%endmacro
|
||||
|
||||
; Overwrites the available slots within the legacy FXSAVE region
|
||||
;
|
||||
; overwrite_xsave_area .xsave_area
|
||||
;
|
||||
; Clobbers RAX
|
||||
;
|
||||
%macro overwrite_fxsave_slots 0
|
||||
; Overwrite the three 16byte "available" slots
|
||||
mov rax, 0x1111111111111111
|
||||
mov qword [rsp + 464 + 8 * 0], rax
|
||||
mov rax, 0x2222222222222222
|
||||
mov qword [rsp + 464 + 8 * 1], rax
|
||||
mov rax, 0x3333333333333333
|
||||
mov qword [rsp + 464 + 8 * 2], rax
|
||||
mov rax, 0x4444444444444444
|
||||
mov qword [rsp + 464 + 8 * 3], rax
|
||||
mov rax, 0x5555555555555555
|
||||
mov qword [rsp + 464 + 8 * 4], rax
|
||||
mov rax, 0x6666666666666666
|
||||
mov qword [rsp + 464 + 8 * 5], rax
|
||||
%endmacro
|
||||
|
||||
; Overwrites all MM and XMM registers with -1
|
||||
;
|
||||
; Typically used right before an XRSTOR to verify
|
||||
; data is restored properly
|
||||
;
|
||||
; Clobbers RAX
|
||||
;
|
||||
%macro corrupt_mmx_and_xmm_registers 0
|
||||
; Corrupt MMX And XMM state
|
||||
mov rax, -1
|
||||
movq mm0, rax
|
||||
movq mm1, rax
|
||||
movq mm2, rax
|
||||
movq mm3, rax
|
||||
movq mm4, rax
|
||||
movq mm5, rax
|
||||
movq mm6, rax
|
||||
movq mm7, rax
|
||||
|
||||
; Setup XMM state
|
||||
movq xmm0, rax
|
||||
movq xmm1, rax
|
||||
movq xmm2, rax
|
||||
movq xmm3, rax
|
||||
movq xmm4, rax
|
||||
movq xmm5, rax
|
||||
movq xmm6, rax
|
||||
movq xmm7, rax
|
||||
movq xmm8, rax
|
||||
movq xmm9, rax
|
||||
movq xmm10, rax
|
||||
movq xmm11, rax
|
||||
movq xmm12, rax
|
||||
movq xmm13, rax
|
||||
movq xmm14, rax
|
||||
movq xmm15, rax
|
||||
%endmacro
|
||||
|
||||
; At the end of the legacy FXSAVE area, there's three 16-byte regions
|
||||
; available for general purpose use. We re-load these to ensure values
|
||||
; that we put in here via overwrite_xsave_area aren't clobbered.
|
||||
;
|
||||
; Clobbers: RAX, RBX, RCX, RDX, RSI, RDI
|
||||
;
|
||||
%macro load_fxsave_slots 0
|
||||
; Load the three 16 bytes of "available" slots to make sure it wasn't overwritten
|
||||
; Reserved can be overwritten regardless
|
||||
mov rax, qword [rsp + 464 + 8 * 0]
|
||||
mov rbx, qword [rsp + 464 + 8 * 1]
|
||||
mov rcx, qword [rsp + 464 + 8 * 2]
|
||||
mov rdx, qword [rsp + 464 + 8 * 3]
|
||||
mov rsi, qword [rsp + 464 + 8 * 4]
|
||||
mov rdi, qword [rsp + 464 + 8 * 5]
|
||||
%endmacro
|
||||
|
||||
; Defines a region of test data to use
|
||||
%macro define_xmm_data_section 0
|
||||
align 32
|
||||
.xmm_data:
|
||||
dq 0x1112131415161718
|
||||
dq 0xABFDEC3402932039
|
||||
dq 0xA1A2A3A4A5A6A7AA
|
||||
dq 0xABFD392482039840
|
||||
|
||||
dq 0x2122232425262728
|
||||
dq 0xDEFCA93847392992
|
||||
dq 0x4142434445464748
|
||||
dq 0x3987432929293847
|
||||
|
||||
dq 0x3132333435363738
|
||||
dq 0xEADC3284ADCE9339
|
||||
dq 0x6162636465666768
|
||||
dq 0xACDEFACDEFACDEFA
|
||||
|
||||
dq 0x4142434445464748
|
||||
dq 0x3987432929293847
|
||||
dq 0x3132333435363738
|
||||
dq 0xEADC3284ADCE9339
|
||||
|
||||
dq 0x5152535455565758
|
||||
dq 0x3764583402983799
|
||||
dq 0x7172737475767778
|
||||
dq 0x3459238471238023
|
||||
|
||||
dq 0x6162636465666768
|
||||
dq 0xACDEFACDEFACDEFA
|
||||
dq 0xA1AAA3A4A5A6A7A8
|
||||
dq 0x3784769228479192
|
||||
|
||||
dq 0x7172737475767778
|
||||
dq 0x3459238471238023
|
||||
dq 0x6162636465666768
|
||||
dq 0xACDEFACDEFACDEFA
|
||||
|
||||
dq 0x8182838485868788
|
||||
dq 0x9347239480289299
|
||||
dq 0x6162636465666768
|
||||
dq 0xACDEFACDEFACDEFA
|
||||
|
||||
dq 0xCCC2C3C4C5C6C7C8
|
||||
dq 0x3949232903428479
|
||||
dq 0xD1D2D3D4DDD6D7D8
|
||||
dq 0x3674823989ADEF73
|
||||
|
||||
dq 0xA1AAA3A4A5A6A7A8
|
||||
dq 0x3784769228479192
|
||||
dq 0xB1B2B3B4B5B6BBB8
|
||||
dq 0xADEADE3894353499
|
||||
|
||||
dq 0xF1F2FFF4F5F6F7F8
|
||||
dq 0x758734629799389A
|
||||
dq 0xD1D2D3D4DDD6D7D8
|
||||
dq 0x3674823989ADEF73
|
||||
|
||||
dq 0xE1E2E3EEE5E6E7E8
|
||||
dq 0x3756438328472389
|
||||
dq 0xB1B2B3B4B5B6BBB8
|
||||
dq 0xADEADE3894353499
|
||||
|
||||
dq 0xD1D2D3D4DDD6D7D8
|
||||
dq 0x3674823989ADEF73
|
||||
dq 0xA1AAA3A4A5A6A7A8
|
||||
dq 0x3784769228479192
|
||||
|
||||
dq 0xC1C2C3C4C5CCC7C8
|
||||
dq 0xABCDEF3894335820
|
||||
dq 0x6162636465666768
|
||||
dq 0xACDEFACDEFACDEFA
|
||||
|
||||
dq 0xB1B2B3B4B5B6BBB8
|
||||
dq 0xADEADE3894353499
|
||||
dq 0xE1E2E3EEE5E6E7E8
|
||||
dq 0x3756438328472389
|
||||
|
||||
dq 0xA1A2A3A4A5A6A7AA
|
||||
dq 0xABFD392482039840
|
||||
dq 0xB1B2B3B4B5B6BBB8
|
||||
dq 0xADEADE3894353499
|
||||
%endmacro
|
||||
|
||||
%endif
|
||||
@@ -0,0 +1,69 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x1111111111111111",
|
||||
"RBX": "0x2222222222222222",
|
||||
"RCX": "0x3333333333333333",
|
||||
"RDX": "0x4444444444444444",
|
||||
"RSI": "0x5555555555555555",
|
||||
"RDI": "0x6666666666666666",
|
||||
"MM0": "0x1112131415161718",
|
||||
"MM1": "0x2122232425262728",
|
||||
"MM2": "0x3132333435363738",
|
||||
"MM3": "0x4142434445464748",
|
||||
"MM4": "0x5152535455565758",
|
||||
"MM5": "0x6162636465666768",
|
||||
"MM6": "0x7172737475767778",
|
||||
"MM7": "0x8182838485868788",
|
||||
"XMM0": ["0x1112131415161718", "0xABFDEC3402932039"],
|
||||
"XMM1": ["0x2122232425262728", "0xDEFCA93847392992"],
|
||||
"XMM2": ["0x3132333435363738", "0xEADC3284ADCE9339"],
|
||||
"XMM3": ["0x4142434445464748", "0x3987432929293847"],
|
||||
"XMM4": ["0x5152535455565758", "0x3764583402983799"],
|
||||
"XMM5": ["0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM6": ["0x7172737475767778", "0x3459238471238023"],
|
||||
"XMM7": ["0x8182838485868788", "0x9347239480289299"],
|
||||
"XMM8": ["0xCCC2C3C4C5C6C7C8", "0x3949232903428479"],
|
||||
"XMM9": ["0xA1AAA3A4A5A6A7A8", "0x3784769228479192"],
|
||||
"XMM10": ["0xF1F2FFF4F5F6F7F8", "0x758734629799389A"],
|
||||
"XMM11": ["0xE1E2E3EEE5E6E7E8", "0x3756438328472389"],
|
||||
"XMM12": ["0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73"],
|
||||
"XMM13": ["0xC1C2C3C4C5CCC7C8", "0xABCDEF3894335820"],
|
||||
"XMM14": ["0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"],
|
||||
"XMM15": ["0xA1A2A3A4A5A6A7AA", "0xABFD392482039840"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%include "xsave_macros.mac"
|
||||
|
||||
mov rsp, 0xE0000000
|
||||
|
||||
; Set up MMX and XMM state
|
||||
set_up_mmx_state .xmm_data
|
||||
set_up_xmm_state .xmm_data
|
||||
|
||||
overwrite_fxsave_slots
|
||||
|
||||
; Now save our state (X87 and SSE only)
|
||||
mov eax, 0b011
|
||||
xsave [rsp]
|
||||
|
||||
; Corrupt MMX And XMM state
|
||||
corrupt_mmx_and_xmm_registers
|
||||
|
||||
; Now reload the state we just saved
|
||||
xrstor [rsp]
|
||||
|
||||
; Load the three 16bytes of "available" slots to make sure it wasn't overwritten
|
||||
; Reserved can be overwritten regardless
|
||||
load_fxsave_slots
|
||||
|
||||
hlt
|
||||
|
||||
; Give ourselves a region of 1000 bytes set to 0xFF
|
||||
align 64
|
||||
.xsave_data:
|
||||
times 1000 db 0xFF
|
||||
|
||||
define_xmm_data_section
|
||||
@@ -0,0 +1,71 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX"],
|
||||
"RegData": {
|
||||
"RAX": "0x1111111111111111",
|
||||
"RBX": "0x2222222222222222",
|
||||
"RCX": "0x3333333333333333",
|
||||
"RDX": "0x4444444444444444",
|
||||
"RSI": "0x5555555555555555",
|
||||
"RDI": "0x6666666666666666",
|
||||
"MM0": "0x1112131415161718",
|
||||
"MM1": "0x2122232425262728",
|
||||
"MM2": "0x3132333435363738",
|
||||
"MM3": "0x4142434445464748",
|
||||
"MM4": "0x5152535455565758",
|
||||
"MM5": "0x6162636465666768",
|
||||
"MM6": "0x7172737475767778",
|
||||
"MM7": "0x8182838485868788",
|
||||
"XMM0": ["0x1112131415161718", "0xABFDEC3402932039", "0xA1A2A3A4A5A6A7AA", "0xABFD392482039840"],
|
||||
"XMM1": ["0x2122232425262728", "0xDEFCA93847392992", "0x4142434445464748", "0x3987432929293847"],
|
||||
"XMM2": ["0x3132333435363738", "0xEADC3284ADCE9339", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM3": ["0x4142434445464748", "0x3987432929293847", "0x3132333435363738", "0xEADC3284ADCE9339"],
|
||||
"XMM4": ["0x5152535455565758", "0x3764583402983799", "0x7172737475767778", "0x3459238471238023"],
|
||||
"XMM5": ["0x6162636465666768", "0xACDEFACDEFACDEFA", "0xA1AAA3A4A5A6A7A8", "0x3784769228479192"],
|
||||
"XMM6": ["0x7172737475767778", "0x3459238471238023", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM7": ["0x8182838485868788", "0x9347239480289299", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM8": ["0xCCC2C3C4C5C6C7C8", "0x3949232903428479", "0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73"],
|
||||
"XMM9": ["0xA1AAA3A4A5A6A7A8", "0x3784769228479192", "0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"],
|
||||
"XMM10": ["0xF1F2FFF4F5F6F7F8", "0x758734629799389A", "0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73"],
|
||||
"XMM11": ["0xE1E2E3EEE5E6E7E8", "0x3756438328472389", "0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"],
|
||||
"XMM12": ["0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73", "0xA1AAA3A4A5A6A7A8", "0x3784769228479192"],
|
||||
"XMM13": ["0xC1C2C3C4C5CCC7C8", "0xABCDEF3894335820", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM14": ["0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499", "0xE1E2E3EEE5E6E7E8", "0x3756438328472389"],
|
||||
"XMM15": ["0xA1A2A3A4A5A6A7AA", "0xABFD392482039840", "0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%define IS_AVX
|
||||
%include "xsave_macros.mac"
|
||||
|
||||
mov rsp, 0xE0000000
|
||||
|
||||
; Set up MMX and XMM state
|
||||
set_up_mmx_state .xmm_data
|
||||
set_up_xmm_state .xmm_data
|
||||
|
||||
overwrite_fxsave_slots
|
||||
|
||||
; Now save our state (X87, SSE, and AVX only)
|
||||
mov eax, 0b111
|
||||
xsave [rsp]
|
||||
|
||||
; Corrupt MMX And XMM state
|
||||
corrupt_mmx_and_xmm_registers
|
||||
|
||||
; Now reload the state we just saved
|
||||
xrstor [rsp]
|
||||
|
||||
; Load the three 16bytes of "available" slots to make sure it wasn't overwritten
|
||||
; Reserved can be overwritten regardless
|
||||
load_fxsave_slots
|
||||
|
||||
hlt
|
||||
|
||||
; Give ourselves a region of 1000 bytes set to 0xFF
|
||||
align 64
|
||||
.xsave_data:
|
||||
times 1000 db 0xFF
|
||||
|
||||
define_xmm_data_section
|
||||
@@ -0,0 +1,71 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX"],
|
||||
"RegData": {
|
||||
"RAX": "0x1111111111111111",
|
||||
"RBX": "0x2222222222222222",
|
||||
"RCX": "0x3333333333333333",
|
||||
"RDX": "0x4444444444444444",
|
||||
"RSI": "0x5555555555555555",
|
||||
"RDI": "0x6666666666666666",
|
||||
"MM0": "0x1112131415161718",
|
||||
"MM1": "0x2122232425262728",
|
||||
"MM2": "0x3132333435363738",
|
||||
"MM3": "0x4142434445464748",
|
||||
"MM4": "0x5152535455565758",
|
||||
"MM5": "0x6162636465666768",
|
||||
"MM6": "0x7172737475767778",
|
||||
"MM7": "0x8182838485868788",
|
||||
"XMM0": ["0x0000000000000000", "0x0000000000000000", "0xA1A2A3A4A5A6A7AA", "0xABFD392482039840"],
|
||||
"XMM1": ["0x0000000000000000", "0x0000000000000000", "0x4142434445464748", "0x3987432929293847"],
|
||||
"XMM2": ["0x0000000000000000", "0x0000000000000000", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM3": ["0x0000000000000000", "0x0000000000000000", "0x3132333435363738", "0xEADC3284ADCE9339"],
|
||||
"XMM4": ["0x0000000000000000", "0x0000000000000000", "0x7172737475767778", "0x3459238471238023"],
|
||||
"XMM5": ["0x0000000000000000", "0x0000000000000000", "0xA1AAA3A4A5A6A7A8", "0x3784769228479192"],
|
||||
"XMM6": ["0x0000000000000000", "0x0000000000000000", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM7": ["0x0000000000000000", "0x0000000000000000", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM8": ["0x0000000000000000", "0x0000000000000000", "0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73"],
|
||||
"XMM9": ["0x0000000000000000", "0x0000000000000000", "0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"],
|
||||
"XMM10": ["0x0000000000000000", "0x0000000000000000", "0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73"],
|
||||
"XMM11": ["0x0000000000000000", "0x0000000000000000", "0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"],
|
||||
"XMM12": ["0x0000000000000000", "0x0000000000000000", "0xA1AAA3A4A5A6A7A8", "0x3784769228479192"],
|
||||
"XMM13": ["0x0000000000000000", "0x0000000000000000", "0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM14": ["0x0000000000000000", "0x0000000000000000", "0xE1E2E3EEE5E6E7E8", "0x3756438328472389"],
|
||||
"XMM15": ["0x0000000000000000", "0x0000000000000000", "0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%define IS_AVX
|
||||
%include "xsave_macros.mac"
|
||||
|
||||
mov rsp, 0xE0000000
|
||||
|
||||
; Set up MMX and XMM state
|
||||
set_up_mmx_state .xmm_data
|
||||
set_up_xmm_state .xmm_data
|
||||
|
||||
overwrite_fxsave_slots
|
||||
|
||||
; Now save our state (X87 and AVX only)
|
||||
mov eax, 0b101
|
||||
xsave [rsp]
|
||||
|
||||
; Corrupt MMX And XMM state
|
||||
corrupt_mmx_and_xmm_registers
|
||||
|
||||
; Now reload the state we just saved
|
||||
xrstor [rsp]
|
||||
|
||||
; Load the three 16bytes of "available" slots to make sure it wasn't overwritten
|
||||
; Reserved can be overwritten regardless
|
||||
load_fxsave_slots
|
||||
|
||||
hlt
|
||||
|
||||
; Give ourselves a region of 1000 bytes set to 0xFF
|
||||
align 64
|
||||
.xsave_data:
|
||||
times 1000 db 0xFF
|
||||
|
||||
define_xmm_data_section
|
||||
@@ -0,0 +1,69 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x1111111111111111",
|
||||
"RBX": "0x2222222222222222",
|
||||
"RCX": "0x3333333333333333",
|
||||
"RDX": "0x4444444444444444",
|
||||
"RSI": "0x5555555555555555",
|
||||
"RDI": "0x6666666666666666",
|
||||
"MM0": "0",
|
||||
"MM1": "0",
|
||||
"MM2": "0",
|
||||
"MM3": "0",
|
||||
"MM4": "0",
|
||||
"MM5": "0",
|
||||
"MM6": "0",
|
||||
"MM7": "0",
|
||||
"XMM0": ["0x1112131415161718", "0xABFDEC3402932039"],
|
||||
"XMM1": ["0x2122232425262728", "0xDEFCA93847392992"],
|
||||
"XMM2": ["0x3132333435363738", "0xEADC3284ADCE9339"],
|
||||
"XMM3": ["0x4142434445464748", "0x3987432929293847"],
|
||||
"XMM4": ["0x5152535455565758", "0x3764583402983799"],
|
||||
"XMM5": ["0x6162636465666768", "0xACDEFACDEFACDEFA"],
|
||||
"XMM6": ["0x7172737475767778", "0x3459238471238023"],
|
||||
"XMM7": ["0x8182838485868788", "0x9347239480289299"],
|
||||
"XMM8": ["0xCCC2C3C4C5C6C7C8", "0x3949232903428479"],
|
||||
"XMM9": ["0xA1AAA3A4A5A6A7A8", "0x3784769228479192"],
|
||||
"XMM10": ["0xF1F2FFF4F5F6F7F8", "0x758734629799389A"],
|
||||
"XMM11": ["0xE1E2E3EEE5E6E7E8", "0x3756438328472389"],
|
||||
"XMM12": ["0xD1D2D3D4DDD6D7D8", "0x3674823989ADEF73"],
|
||||
"XMM13": ["0xC1C2C3C4C5CCC7C8", "0xABCDEF3894335820"],
|
||||
"XMM14": ["0xB1B2B3B4B5B6BBB8", "0xADEADE3894353499"],
|
||||
"XMM15": ["0xA1A2A3A4A5A6A7AA", "0xABFD392482039840"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%include "xsave_macros.mac"
|
||||
|
||||
mov rsp, 0xE0000000
|
||||
|
||||
; Set up MMX and XMM state
|
||||
set_up_mmx_state .xmm_data
|
||||
set_up_xmm_state .xmm_data
|
||||
|
||||
overwrite_fxsave_slots
|
||||
|
||||
; Now save our state (SSE only)
|
||||
mov eax, 0b010
|
||||
xsave [rsp]
|
||||
|
||||
; Corrupt MMX And XMM state
|
||||
corrupt_mmx_and_xmm_registers
|
||||
|
||||
; Now reload the state we just saved
|
||||
xrstor [rsp]
|
||||
|
||||
; Load the three 16bytes of "available" slots to make sure it wasn't overwritten
|
||||
; Reserved can be overwritten regardless
|
||||
load_fxsave_slots
|
||||
|
||||
hlt
|
||||
|
||||
; Give ourselves a region of 1000 bytes set to 0xFF
|
||||
align 64
|
||||
.xsave_data:
|
||||
times 1000 db 0xFF
|
||||
|
||||
define_xmm_data_section
|
||||
@@ -0,0 +1,69 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x1111111111111111",
|
||||
"RBX": "0x2222222222222222",
|
||||
"RCX": "0x3333333333333333",
|
||||
"RDX": "0x4444444444444444",
|
||||
"RSI": "0x5555555555555555",
|
||||
"RDI": "0x6666666666666666",
|
||||
"MM0": "0x1112131415161718",
|
||||
"MM1": "0x2122232425262728",
|
||||
"MM2": "0x3132333435363738",
|
||||
"MM3": "0x4142434445464748",
|
||||
"MM4": "0x5152535455565758",
|
||||
"MM5": "0x6162636465666768",
|
||||
"MM6": "0x7172737475767778",
|
||||
"MM7": "0x8182838485868788",
|
||||
"XMM0": ["0", "0"],
|
||||
"XMM1": ["0", "0"],
|
||||
"XMM2": ["0", "0"],
|
||||
"XMM3": ["0", "0"],
|
||||
"XMM4": ["0", "0"],
|
||||
"XMM5": ["0", "0"],
|
||||
"XMM6": ["0", "0"],
|
||||
"XMM7": ["0", "0"],
|
||||
"XMM8": ["0", "0"],
|
||||
"XMM9": ["0", "0"],
|
||||
"XMM10": ["0", "0"],
|
||||
"XMM11": ["0", "0"],
|
||||
"XMM12": ["0", "0"],
|
||||
"XMM13": ["0", "0"],
|
||||
"XMM14": ["0", "0"],
|
||||
"XMM15": ["0", "0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%include "xsave_macros.mac"
|
||||
|
||||
mov rsp, 0xE0000000
|
||||
|
||||
; Set up MMX and XMM state
|
||||
set_up_mmx_state .xmm_data
|
||||
set_up_xmm_state .xmm_data
|
||||
|
||||
overwrite_fxsave_slots
|
||||
|
||||
; Now save our state (X87 only)
|
||||
mov eax, 0b001
|
||||
xsave [rsp]
|
||||
|
||||
; Corrupt MMX And XMM state
|
||||
corrupt_mmx_and_xmm_registers
|
||||
|
||||
; Now reload the state we just saved
|
||||
xrstor [rsp]
|
||||
|
||||
; Load the three 16bytes of "available" slots to make sure it wasn't overwritten
|
||||
; Reserved can be overwritten regardless
|
||||
load_fxsave_slots
|
||||
|
||||
hlt
|
||||
|
||||
; Give ourselves a region of 1000 bytes set to 0xFF
|
||||
align 64
|
||||
.xsave_data:
|
||||
times 1000 db 0xFF
|
||||
|
||||
define_xmm_data_section
|
||||
Reference in new issue
Block a user