mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 15:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cc85a6a722 | ||
|
|
e72fa02897 | ||
|
|
8a4c5bcc65 | ||
|
|
7a13a24c05 | ||
|
|
5a53931b92 | ||
|
|
f9b352a093 | ||
|
|
f444b03317 | ||
|
|
ed05846dd0 | ||
|
|
f609990f90 | ||
|
|
d2032da452 | ||
|
|
8047007a7a | ||
|
|
1f7e82ea09 | ||
|
|
35c52f20f9 | ||
|
|
17c82c22a6 | ||
|
|
df03a7b101 | ||
|
|
c859540d7e | ||
|
|
20794593e7 | ||
|
|
a80a2bf569 | ||
|
|
e86a792189 | ||
|
|
d6c9b549df | ||
|
|
1a4d5a1abb | ||
|
|
c3e123df25 | ||
|
|
cac798574a | ||
|
|
1506a19229 | ||
|
|
51861234bc | ||
|
|
677b72c9a5 | ||
|
|
71a8c66c95 | ||
|
|
3372e9bdbb | ||
|
|
df0723e14b | ||
|
|
7ee6fc0d7f | ||
|
|
bf773452ac | ||
|
|
95dbccc0ab | ||
|
|
e5189d63a2 | ||
|
|
7d5442357a | ||
|
|
9dcc1deec0 | ||
|
|
66d4206cd7 | ||
|
|
f39163b1e1 | ||
|
|
01837b3ad6 | ||
|
|
bdb68840e3 | ||
|
|
4e2dcf3298 | ||
|
|
628f825416 | ||
|
|
a80327f6df | ||
|
|
16f7002222 | ||
|
|
c9712e45cb | ||
|
|
a082161d72 | ||
|
|
9017325c95 | ||
|
|
ae536e44d7 | ||
|
|
7679485cc3 | ||
|
|
cb8bf1add6 | ||
|
|
a69c457715 | ||
|
|
7c4729678b | ||
|
|
537562fab7 | ||
|
|
f8721992c2 | ||
|
|
e652399fd3 | ||
|
|
e7c92c43a0 | ||
|
|
9b5e1c44c8 | ||
|
|
755600c371 | ||
|
|
bec8b70e5d | ||
|
|
bef8ddde48 | ||
|
|
fe06f1b151 | ||
|
|
92a15e00c7 | ||
|
|
2997257d6d | ||
|
|
7ceadc6b5b | ||
|
|
8c41e8f7d8 | ||
|
|
784b3064fc | ||
|
|
b3bc1e23cc | ||
|
|
1a9b6a89f4 | ||
|
|
21bf35d211 | ||
|
|
9473025b18 | ||
|
|
02f15f4099 | ||
|
|
e007789ced | ||
|
|
a2b165043c | ||
|
|
5b5808218b | ||
|
|
41ec987f3e | ||
|
|
0f4a5edf4f | ||
|
|
03f73531d3 | ||
|
|
69181d438c | ||
|
|
a2cbfccb3b | ||
|
|
4e01452a65 | ||
|
|
cc7a56b1a6 | ||
|
|
0b0dd3891e | ||
|
|
8bc33e95c1 | ||
|
|
96a0364a86 | ||
|
|
c0a783997d | ||
|
|
f78537109d | ||
|
|
0c156ed6f9 | ||
|
|
920913cf80 | ||
|
|
8840b2154c | ||
|
|
e02be8073e | ||
|
|
f75d3550b4 | ||
|
|
802c588695 | ||
|
|
fd962f40d7 | ||
|
|
a9b660af69 | ||
|
|
fd5c36ba9c | ||
|
|
5be798e9e6 | ||
|
|
09997cff9c | ||
|
|
c9d1f0d75a | ||
|
|
95b7592241 | ||
|
|
1dc4f8c429 | ||
|
|
1d7fcdb54a | ||
|
|
45d3b83143 | ||
|
|
d97fa9af14 | ||
|
|
c9101d3f68 | ||
|
|
52f64a0c7b | ||
|
|
737f917838 | ||
|
|
a6c6248bcb | ||
|
|
5646428640 | ||
|
|
0c8df2beaf | ||
|
|
de0f3984e9 | ||
|
|
ada226bbb4 | ||
|
|
4bc5a09e62 | ||
|
|
5704b5f23f | ||
|
|
6017a9135a | ||
|
|
6ef6d9c391 | ||
|
|
0ad6f98a8c | ||
|
|
1354f92cc5 | ||
|
|
8b90caad95 | ||
|
|
3a4a965347 | ||
|
|
45cdab2ac3 | ||
|
|
b89dc56ae1 | ||
|
|
d675b4af6f | ||
|
|
d75fb38344 | ||
|
|
4cb385a27b | ||
|
|
9a4fdd8059 | ||
|
|
363411f0c7 | ||
|
|
5bc418407c | ||
|
|
46e2dc7498 | ||
|
|
4a54197868 | ||
|
|
3f214dd244 | ||
|
|
61ca651fe1 | ||
|
|
cd0a340d29 | ||
|
|
77e8be1215 | ||
|
|
182010ca97 | ||
|
|
f7c663240e | ||
|
|
e9244680aa | ||
|
|
82b4aef30d | ||
|
|
22919a5b65 | ||
|
|
00dc373bb9 | ||
|
|
e593237670 | ||
|
|
0a4bf10ba5 | ||
|
|
f47caf48c6 | ||
|
|
5674d3a871 | ||
|
|
fde64aedf7 | ||
|
|
e03b859c20 | ||
|
|
ce4e991a6e | ||
|
|
7d822ba1c8 | ||
|
|
f90dcd2eb1 | ||
|
|
adbdd33ece | ||
|
|
06250d806d | ||
|
|
613ed559e7 | ||
|
|
8ac3841946 | ||
|
|
458259bf47 | ||
|
|
dc65a5ef8c | ||
|
|
2fc529d5b7 | ||
|
|
ea489567da | ||
|
|
ed69eb9f6f | ||
|
|
6eae064511 |
No files matched your search
@@ -25,7 +25,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -241,13 +241,19 @@ jobs:
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -33,7 +33,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -176,13 +176,19 @@ jobs:
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -111,7 +111,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
+6
-1
@@ -7,6 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
@@ -18,7 +19,6 @@ option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
@@ -180,8 +180,13 @@ endif()
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
set(BUILD_FEXCONFIG FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
|
||||
@@ -282,6 +282,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tIROps Op;\n\n")
|
||||
output_file.write("\tuint8_t Size;\n")
|
||||
output_file.write("\tuint8_t ElementSize;\n")
|
||||
output_file.write("\tuint8_t _pad;\n")
|
||||
|
||||
output_file.write("\ttemplate<typename T>\n")
|
||||
output_file.write("\tT const* C() const { return reinterpret_cast<T const*>(Data); }\n")
|
||||
@@ -291,6 +292,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tOrderedNodeWrapper Args[0];\n")
|
||||
|
||||
output_file.write("};\n\n");
|
||||
output_file.write("static_assert(sizeof(IROp_Header) == sizeof(uint32_t), \"IROp_Header should be 32-bits in size\");\n\n");
|
||||
|
||||
# Now the user defined types
|
||||
output_file.write("// User defined IR Op structs\n")
|
||||
|
||||
+1
-1
@@ -231,7 +231,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl xxhash tiny-json FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
|
||||
+5
-1
@@ -17,8 +17,12 @@ namespace FEXCore {
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
#endif
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
|
||||
-216
@@ -27,8 +27,6 @@
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
}
|
||||
@@ -42,71 +40,6 @@ namespace DefaultValues {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
@@ -583,154 +516,5 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
+4
-35
@@ -66,10 +66,6 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::AddVirtualMemoryMapping([[maybe_unused]] uint64_t VirtualAddress, [[maybe_unused]] uint64_t PhysicalAddress, [[maybe_unused]] uint64_t Size) {
|
||||
return false;
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
@@ -87,38 +83,11 @@ namespace FEXCore::Context {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
|
||||
return CPUID.RunXCRFunction(Function);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
//void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
// CTX->CompileRIP(CTX->ParentThread, RIP);
|
||||
//}
|
||||
//uint64_t GetThreadCount(FEXCore::Context::Context *CTX) {
|
||||
// return CTX->GetThreadCount();
|
||||
//}
|
||||
|
||||
//FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(FEXCore::Context::Context *CTX, uint64_t Thread) {
|
||||
// return CTX->GetRuntimeStatsForThread(Thread);
|
||||
//}
|
||||
|
||||
//bool GetDebugDataForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
// return CTX->GetDebugDataForRIP(RIP, Data);
|
||||
//}
|
||||
|
||||
//bool FindHostCodeForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, uint8_t **Code) {
|
||||
// return CTX->FindHostCodeForRIP(RIP, Code);
|
||||
//}
|
||||
|
||||
// XXX:
|
||||
// bool FindIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir) {
|
||||
// return CTX->FindIRForRIP(RIP, ir);
|
||||
// }
|
||||
|
||||
// void SetIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir) {
|
||||
// CTX->SetIRForRIP(RIP, ir);
|
||||
// }
|
||||
}
|
||||
|
||||
}
|
||||
+36
-17
@@ -1,7 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
@@ -14,6 +13,7 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
@@ -100,8 +100,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
@@ -154,11 +152,14 @@ namespace FEXCore::Context {
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
@@ -180,8 +181,8 @@ namespace FEXCore::Context {
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
@@ -253,7 +254,7 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
@@ -289,7 +290,8 @@ namespace FEXCore::Context {
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(static_cast<ContextImpl*>(Frame->Thread->CTX)->CodeInvalidationMutex);
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -300,20 +302,13 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
uint64_t GetThreadCount() const;
|
||||
FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(uint64_t Thread);
|
||||
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
@@ -372,11 +367,32 @@ namespace FEXCore::Context {
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override { ExitOnHLT = true; }
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
}
|
||||
else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
@@ -411,6 +427,9 @@ namespace FEXCore::Context {
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
|
||||
+197
-79
@@ -20,6 +20,131 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
}
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
{FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
@@ -29,22 +154,21 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
ConfiguredGPRs = NumGPRs64;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs64;
|
||||
ConfiguredGPRPairs = NumGPRPairs64;
|
||||
ConfiguredFPRs = NumFPRs64;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs64;
|
||||
ConfiguredDynamicGPRs = NumGPRs64 - NumGPRs64; // Will be zero, just to be consistent with 32-bit side
|
||||
ConfiguredDynamicRegisterBase = nullptr;
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredGPRs = NumGPRs32;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs32;
|
||||
ConfiguredGPRPairs = NumGPRPairs32;
|
||||
ConfiguredFPRs = NumFPRs32;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs32;
|
||||
ConfiguredDynamicGPRs = NumGPRs32 - NumGPRs64; // Will be 8
|
||||
ConfiguredDynamicRegisterBase = &RA64[9];
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralPairRegisters = x32::RAPair;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -219,14 +343,14 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -241,8 +365,8 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
@@ -252,22 +376,20 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
} else {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRSpillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
@@ -299,8 +421,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TMP4.R());
|
||||
@@ -310,22 +432,22 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRFillMask)];
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
@@ -342,9 +464,9 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -360,10 +482,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicGPRs + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = ConfiguredFPRs * FPRRegSize;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -372,31 +494,29 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(ConfiguredFPRs % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
@@ -406,30 +526,28 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
@@ -26,87 +26,9 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9 + 8> RA64 = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4 + 3> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
// Registers only available on 32-bit
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17}
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12 + 8> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
@@ -139,32 +61,16 @@ protected:
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
uint32_t ConfiguredGPRs;
|
||||
uint32_t ConfiguredSRAGPRs;
|
||||
uint32_t ConfiguredGPRPairs;
|
||||
uint32_t ConfiguredFPRs;
|
||||
uint32_t ConfiguredSRAFPRs;
|
||||
uint32_t ConfiguredDynamicGPRs;
|
||||
const FEXCore::ARMEmitter::Register *ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
// 64-bit gets removal of additional pairs
|
||||
constexpr static uint32_t NumGPRs64 = RA64.size() - 8;
|
||||
constexpr static uint32_t NumSRAGPRs64 = SRA64.size();
|
||||
constexpr static uint32_t NumFPRs64 = RAFPR.size() - 8;
|
||||
constexpr static uint32_t NumSRAFPRs64 = SRAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs64 = RA64Pair.size() - 3;
|
||||
|
||||
// 32-bit gets full array of GPR registers
|
||||
// SRA registers remove the additional 8
|
||||
constexpr static uint32_t NumGPRs32 = RA64.size();
|
||||
constexpr static uint32_t NumSRAGPRs32 = SRA64.size() - 8;
|
||||
constexpr static uint32_t NumFPRs32 = RAFPR.size();
|
||||
constexpr static uint32_t NumSRAFPRs32 = SRAFPR.size() - 8;
|
||||
constexpr static uint32_t NumGPRPairs32 = RA64Pair.size();
|
||||
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
@@ -183,7 +89,7 @@ protected:
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
|
||||
@@ -145,6 +145,27 @@ public:
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0000, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0001, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0010, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0011, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
@@ -381,6 +402,26 @@ public:
|
||||
(0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
@@ -467,7 +508,24 @@ public:
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
void ctz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
// TODO: PAUTH
|
||||
|
||||
@@ -819,6 +877,21 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void MinMaxImmediate(uint32_t opc, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = 0b1'0001'11U << 22;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= opc << 18;
|
||||
Instr |= (Imm & 0xFF) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Move Wide
|
||||
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -899,6 +972,9 @@ private:
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == FEXCore::ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
|
||||
+25
-16
@@ -76,10 +76,6 @@ static uint32_t GetCPUID() {
|
||||
return CPU;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
0 | // Stepping
|
||||
@@ -399,8 +395,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -433,15 +427,15 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(CTX->HostFeatures.SupportsCRC << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
(0 << 24) | // APIC TSC-Deadline
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(SupportsAVX() << 26) | // XSAVE
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
@@ -630,12 +624,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(1 << 3) | // BMI1
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(1 << 8) | // BMI2
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
@@ -736,13 +730,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
// Leaf 0
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
uint32_t XFeatureSupportedSizeMax = SUPPORTS_AVX ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
uint32_t XFeatureSupportedSizeMax = SupportsAVX() ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
if (Leaf == 0) {
|
||||
// XFeatureSupportedMask[31:0]
|
||||
Res.eax =
|
||||
(1 << 0) | // X87 support
|
||||
(1 << 1) | // 128-bit SSE support
|
||||
(SUPPORTS_AVX << 2) | // 256-bit AVX support
|
||||
(SupportsAVX() << 2) | // 256-bit AVX support
|
||||
(0b00 << 3) | // MPX State
|
||||
(0b000 << 5) | // AVX-512 state
|
||||
(0 << 8) | // "Used for IA32_XSS" ... Used for what?
|
||||
@@ -776,8 +770,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
Res.edx = 0;
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
Res.eax = SUPPORTS_AVX ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SUPPORTS_AVX ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
Res.eax = SupportsAVX() ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SupportsAVX() ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
|
||||
// Reserved
|
||||
Res.ecx = 0;
|
||||
@@ -1212,11 +1206,26 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() {
|
||||
// This just returns XCR0
|
||||
FEXCore::CPUID::XCRResults Res{
|
||||
.eax = static_cast<uint32_t>(XCR0),
|
||||
.edx = static_cast<uint32_t>(XCR0 >> 32),
|
||||
};
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
CTX = ctx;
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+41
@@ -63,13 +63,52 @@ public:
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) {
|
||||
if (Function >= 1) {
|
||||
// XCR function 1 is not yet supported.
|
||||
return {};
|
||||
}
|
||||
|
||||
return XCRFunction_0h();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
// Affects XSAVE and XRSTOR when modified.
|
||||
// Bit layout is as follows.
|
||||
// [0] - x87 enabled
|
||||
// [1] - SSE enabled
|
||||
// [2] - YMM enabled (256-bit SSE)
|
||||
// [8:3] - Reserved. MBZ.
|
||||
// [9] - MPK
|
||||
// [10] - Reserved. MBZ.
|
||||
// [11] - CET_U
|
||||
// [12] - CET_S
|
||||
// [61:13] - Reserved. MBZ.
|
||||
// [62] - LWP (Lightweight profiling)
|
||||
// [63] - Reserved for XCR bit vector expansion. MBZ.
|
||||
// Always enable x87 and SSE by default.
|
||||
constexpr static uint64_t XCR0_X87 = 1ULL << 0;
|
||||
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
|
||||
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
|
||||
|
||||
uint64_t XCR0 {
|
||||
XCR0_X87 |
|
||||
XCR0_SSE
|
||||
};
|
||||
|
||||
uint32_t SupportsAVX() const {
|
||||
return (XCR0 & XCR0_AVX) ? 1 : 0;
|
||||
}
|
||||
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
@@ -109,6 +148,8 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h();
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
static constexpr std::array<FunctionHandler, 27> Primary = {
|
||||
// 0: Highest function parameter and ID
|
||||
|
||||
+90
-106
@@ -8,6 +8,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
@@ -24,6 +25,8 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
@@ -42,6 +45,7 @@ $end_info$
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
@@ -162,6 +166,9 @@ namespace FEXCore::Context {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
@@ -187,43 +194,38 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
static FEXCore::Core::CPUState CreateDefaultCPUState() {
|
||||
FEXCore::Core::CPUState NewThreadState{};
|
||||
|
||||
// Initialize default CPU state
|
||||
NewThreadState.rip = ~0ULL;
|
||||
for (auto& greg : NewThreadState.gregs) {
|
||||
greg = 0;
|
||||
}
|
||||
|
||||
for (auto& xmm : NewThreadState.xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(NewThreadState.flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
NewThreadState.flags[1] = 1;
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
NewThreadState.FTW = 0xFFFF;
|
||||
return NewThreadState;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
const CPU::CPUBackend::JITCodeHeader *InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
const CPU::CPUBackend::JITCodeTail *InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
|
||||
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
return InlineTail->RIP;
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto &RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
}
|
||||
else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
}
|
||||
}
|
||||
return StartingGuestRIP;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -304,12 +306,13 @@ namespace FEXCore::Context {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0);
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
@@ -552,7 +555,9 @@ namespace FEXCore::Context {
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -616,7 +621,9 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
@@ -625,6 +632,9 @@ namespace FEXCore::Context {
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
|
||||
// Insert after the Thread object has been fully initialized
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
@@ -650,10 +660,22 @@ namespace FEXCore::Context {
|
||||
// To be able to delete a thread from itself, we need to detached the std::thread object
|
||||
Thread->ExecutionThread->detach();
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::CleanupAfterFork(FEXCore::Core::InternalThreadState *LiveThread) {
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState *LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto &DeadThread : Threads) {
|
||||
@@ -692,6 +714,11 @@ namespace FEXCore::Context {
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
}
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
@@ -712,36 +739,30 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
#ifndef _WIN32
|
||||
int FD {-1};
|
||||
bool CloseAfter = false;
|
||||
FEXCore::File::File FD;
|
||||
const auto DumpIRStr = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR();
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpIRStr =="stderr" || DumpIRStr =="no") {
|
||||
FD = STDERR_FILENO;
|
||||
FD = FEXCore::File::File::GetStdERR();
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
FD = STDOUT_FILENO;
|
||||
FD = FEXCore::File::File::GetStdOUT();
|
||||
}
|
||||
else {
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
FD = open(fileName.c_str(), O_CREAT | O_WRONLY | O_TRUNC | O_CLOEXEC, USER_PERMS, USER_PERMS);
|
||||
CloseAfter = true;
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
}
|
||||
|
||||
if (FD != -1) {
|
||||
if (FD.IsValid()) {
|
||||
fextl::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
close(FD);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
static void ValidateIR(ContextImpl *ctx, IR::IREmitter *IREmitter) {
|
||||
@@ -795,7 +816,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
}
|
||||
});
|
||||
|
||||
@@ -825,7 +846,7 @@ namespace FEXCore::Context {
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
if (ExtendedDebugInfo) {
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
|
||||
@@ -974,7 +995,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
@@ -983,7 +1004,7 @@ namespace FEXCore::Context {
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -1006,9 +1027,6 @@ namespace FEXCore::Context {
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
// Increment stats
|
||||
Thread->Stats.BlocksCompiled.fetch_add(1);
|
||||
|
||||
// These blocks aren't already in the cache
|
||||
GeneratedIR = true;
|
||||
}
|
||||
@@ -1047,7 +1065,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
std::shared_lock lk(CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1079,7 +1097,7 @@ namespace FEXCore::Context {
|
||||
auto FragmentBasePtr = reinterpret_cast<uint8_t *>(CodePtr);
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
@@ -1147,6 +1165,7 @@ namespace FEXCore::Context {
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
@@ -1191,6 +1210,7 @@ namespace FEXCore::Context {
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
@@ -1221,14 +1241,20 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex);
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex);
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1237,6 +1263,7 @@ namespace FEXCore::Context {
|
||||
void ContextImpl::MarkMemoryShared() {
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
@@ -1255,7 +1282,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::shared_lock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
@@ -1290,56 +1317,11 @@ namespace FEXCore::Context {
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidateGuestCodeRange(Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
InvalidateGuestCodeRange(nullptr, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
});
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
void Context::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
uint64_t RIPBackup = Thread->CurrentFrame->State.rip;
|
||||
Thread->CurrentFrame->State.rip = RIP;
|
||||
|
||||
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
CTX->ThreadRemoveCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CTX->CompileBlock(Thread->CurrentFrame, RIP);
|
||||
|
||||
Thread->CurrentFrame->State.rip = RIPBackup;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::GetThreadCount() const {
|
||||
return Threads.size();
|
||||
}
|
||||
|
||||
FEXCore::Core::RuntimeStats *ContextImpl::GetRuntimeStatsForThread(uint64_t Thread) {
|
||||
return &Threads[Thread]->Stats;
|
||||
}
|
||||
|
||||
bool ContextImpl::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
std::lock_guard<std::recursive_mutex> lk(ParentThread->LookupCache->WriteLock);
|
||||
auto it = ParentThread->DebugStore.find(RIP);
|
||||
if (it == ParentThread->DebugStore.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(Data, it->second.DebugData.get(), sizeof(FEXCore::Core::DebugData));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ContextImpl::FindHostCodeForRIP(uint64_t RIP, uint8_t **Code) {
|
||||
uintptr_t HostCode = ParentThread->LookupCache->FindBlock(RIP);
|
||||
if (!HostCode) {
|
||||
return false;
|
||||
}
|
||||
|
||||
*Code = reinterpret_cast<uint8_t*>(HostCode);
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) {
|
||||
uint64_t Result{};
|
||||
Result = Handler->HandleSyscall(Frame, Args);
|
||||
@@ -1362,7 +1344,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
|
||||
@@ -129,13 +129,6 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
@@ -183,7 +176,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -197,28 +190,11 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
#endif
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
@@ -230,28 +206,17 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(ARMEmitter::XReg::x4, ARMEmitter::XReg::x0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
mov(ARMEmitter::XReg::x0, ARMEmitter::XReg::x4);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
|
||||
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
@@ -260,31 +225,11 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
Bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, 8);
|
||||
}
|
||||
#endif
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
@@ -297,25 +242,17 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP1, 0);
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
@@ -341,7 +278,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
@@ -352,7 +289,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
brk(0);
|
||||
}
|
||||
@@ -363,20 +300,28 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
if (CTX->ExitOnHLTEnabled()) {
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
|
||||
PopCalleeSavedRegisters();
|
||||
ret();
|
||||
}
|
||||
else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
@@ -453,7 +398,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -475,7 +420,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -497,7 +442,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -519,7 +464,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
|
||||
@@ -31,22 +31,22 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
void EmitDispatcher();
|
||||
|
||||
uint16_t GetSRAGPRCount() const override {
|
||||
return SRA64.size();
|
||||
return StaticRegisters.size();
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const override {
|
||||
return SRAFPR.size();
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRA64.size(); ++i) {
|
||||
Mapping[i] = SRA64[i].Idx();
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRAFPR.size(); ++i) {
|
||||
Mapping[i] = SRAFPR[i].Idx();
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -90,6 +90,8 @@ public:
|
||||
virtual void GetSRAFPRMapping(uint8_t Mapping[16]) const {
|
||||
}
|
||||
|
||||
const DispatcherConfig& GetConfig() const { return config; }
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
|
||||
@@ -170,38 +170,11 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
#endif
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
@@ -210,26 +183,15 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(rax);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rax, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(rax, qword [rax]);
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
#endif
|
||||
L(AfterStore);
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
@@ -237,34 +199,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
#endif
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
@@ -272,30 +207,17 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(qword [rbx], rbx);
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
L(AfterStore);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
jmp(rax);
|
||||
}
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
|
||||
@@ -54,7 +54,12 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
#ifndef _WIN32
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
// Need to use ID registers in WINE.
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
|
||||
@@ -143,6 +143,15 @@ DEF_OP(CPUID) {
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
uint32_t *DstPtr = GetDest<uint32_t*>(Data->SSAData, Node);
|
||||
const uint32_t Function = *GetSrc<uint32_t*>(Data->SSAData, Op->Function);
|
||||
|
||||
auto Results = Data->State->CTX->RunXCRFunction(Function);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+1
-1
@@ -46,7 +46,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const uint32_t upper_limit = (16U >> (control & 1)) - 1;
|
||||
const int32_t upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
|
||||
@@ -118,6 +118,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
|
||||
@@ -154,6 +154,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
|
||||
@@ -2280,16 +2280,13 @@ DEF_OP(VRev64) {
|
||||
|
||||
DEF_OP(VPCMPESTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto RAX = *GetSrc<uint64_t*>(Data->SSAData, Op->RAX);
|
||||
const auto RDX = *GetSrc<uint64_t*>(Data->SSAData, Op->RDX);
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPESTRX>::handle(RAX, RDX, LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
|
||||
@@ -477,9 +477,9 @@ DEF_OP(PDep) {
|
||||
const auto IndexReg = TMP4.R();
|
||||
const auto ZeroReg = ARMEmitter::Reg::zr;
|
||||
|
||||
const auto InputReg = SRA64[0];
|
||||
const auto MaskReg = SRA64[1];
|
||||
const auto DestReg = SRA64[2];
|
||||
const auto InputReg = StaticRegisters[0];
|
||||
const auto MaskReg = StaticRegisters[1];
|
||||
const auto DestReg = StaticRegisters[2];
|
||||
|
||||
const auto SpillCode = 1U << InputReg.Idx() |
|
||||
1U << MaskReg.Idx() |
|
||||
@@ -494,7 +494,7 @@ DEF_OP(PDep) {
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, SpillCode);
|
||||
SpillStaticRegs(TMP1, false, SpillCode);
|
||||
|
||||
|
||||
mov(EmitSize, InputReg, Input);
|
||||
@@ -558,7 +558,7 @@ DEF_OP(PExt) {
|
||||
// We sadly need to spill a reg for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, 1U << Mask.Idx());
|
||||
SpillStaticRegs(TMP2, false, 1U << Mask.Idx());
|
||||
mov(EmitSize, Mask, ZeroReg);
|
||||
|
||||
// Main loop
|
||||
|
||||
@@ -22,7 +22,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
@@ -175,7 +175,7 @@ DEF_OP(Syscall) {
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(true, GPRSpillMask, FPRSpillMask);
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -216,8 +216,11 @@ DEF_OP(Syscall) {
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Move result to its destination register
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -257,7 +260,7 @@ DEF_OP(InlineSyscall) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -325,7 +328,7 @@ DEF_OP(Thunk) {
|
||||
// X0: CTX
|
||||
// X1: Args (from guest stack)
|
||||
|
||||
SpillStaticRegs(); // spill to ctx before ra64 spill
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -400,12 +403,12 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X1: RIP
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
@@ -421,7 +424,7 @@ DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
@@ -447,6 +450,34 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
+74
-44
@@ -83,7 +83,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -103,7 +103,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F32:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
@@ -127,7 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -153,7 +153,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -183,7 +183,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -209,7 +209,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -235,7 +235,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -259,7 +259,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -285,7 +285,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -310,7 +310,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -335,7 +335,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -360,7 +360,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -388,7 +388,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -415,7 +415,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -446,20 +446,17 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
|
||||
@@ -486,7 +483,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
}
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
@@ -587,14 +584,14 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, ConfiguredGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, ConfiguredSRAGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, ConfiguredFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, ConfiguredSRAFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, ConfiguredGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, GeneralPairRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < ConfiguredGPRPairs; ++i) {
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
@@ -615,6 +612,11 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
@@ -634,6 +636,25 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (CTX->Dispatcher->GetConfig().StaticRegisterAllocation) {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegisterSRA;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegisterSRA;
|
||||
}
|
||||
else {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegister;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegister;
|
||||
}
|
||||
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
}
|
||||
else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -814,6 +835,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
@@ -891,6 +913,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
@@ -920,8 +943,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
@@ -930,22 +953,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
case FEXCore::IR::IROps::OP_LOADMEMTSO:
|
||||
if (ParanoidTSO()) {
|
||||
Op_ParanoidLoadMemTSO(IROp, ID);
|
||||
}
|
||||
else {
|
||||
Op_LoadMemTSO(IROp, ID);
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::IROps::OP_STOREMEMTSO:
|
||||
if (ParanoidTSO()) {
|
||||
Op_ParanoidStoreMemTSO(IROp, ID);
|
||||
}
|
||||
else {
|
||||
Op_StoreMemTSO(IROp, ID);
|
||||
}
|
||||
break;
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
|
||||
@@ -1093,12 +1102,33 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
+18
-5
@@ -70,9 +70,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -84,9 +84,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -97,7 +97,7 @@ private:
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return RA64Pair[Reg.Reg];
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
@@ -210,6 +210,16 @@ private:
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
OpType RT_StoreRegister;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
@@ -297,6 +307,7 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -318,6 +329,8 @@ private:
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
|
||||
+280
-49
@@ -128,13 +128,177 @@ DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ldrb(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldrh(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).W(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrb(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrh(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 8: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
strb(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
strh(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.W(), STATE, Op->Offset);
|
||||
break;
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs: {}", regOffs);
|
||||
strh(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -170,9 +334,9 @@ DEF_OP(LoadRegister) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -312,7 +476,7 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
DEF_OP(StoreRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -320,9 +484,9 @@ DEF_OP(StoreRegister) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -356,9 +520,9 @@ DEF_OP(StoreRegister) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -500,7 +664,6 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1097,7 +1260,6 @@ DEF_OP(LoadMemTSO) {
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -1860,32 +2022,78 @@ DEF_OP(MemCpy) {
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldapr(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldapr(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(Dst, Addr);
|
||||
ldarb(Dst, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
ldarh(Dst, Addr);
|
||||
ldarh(Dst, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldar(Dst.W(), Addr);
|
||||
ldar(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldar(Dst.X(), Addr);
|
||||
ldar(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
@@ -1896,31 +2104,30 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i8Bit, Dst, 0, TMP1);
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
ldarh(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 0, TMP1);
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
ldar(TMP1.W(), Addr);
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 8:
|
||||
ldar(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case 16:
|
||||
nop();
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, Addr);
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), Addr);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default:
|
||||
@@ -1934,26 +2141,50 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
stlurh(Src, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
stlur(Src.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
stlur(Src.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
stlrb(Src, Addr);
|
||||
stlrb(Src, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
stlrh(Src, Addr);
|
||||
stlrh(Src, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
stlr(Src.W(), Addr);
|
||||
stlr(Src.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
stlr(Src.X(), Addr);
|
||||
stlr(Src.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
@@ -1966,19 +2197,19 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, Addr);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, Addr);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), Addr);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, Addr);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case 16: {
|
||||
// Move vector to GPRs
|
||||
@@ -1988,14 +2219,14 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, Addr); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, Addr); // <- Can also hit SIGBUS
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, Addr, 0);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -12,6 +12,8 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
@@ -37,7 +39,6 @@ DEF_OP(Fence) {
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
@@ -59,15 +60,15 @@ DEF_OP(Break) {
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
case Core::FAULT_SIGILL:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
case Core::FAULT_SIGTRAP:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
case Core::FAULT_SIGSEGV:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
|
||||
br(TMP1);
|
||||
break;
|
||||
@@ -77,11 +78,6 @@ DEF_OP(Break) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
#else
|
||||
DEF_OP(Break) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
auto Dst = GetReg(Node);
|
||||
@@ -148,7 +144,7 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
@@ -174,7 +170,7 @@ DEF_OP(ProcessorID) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
|
||||
@@ -146,6 +146,7 @@ DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// XXX: This is very terrible, but I don't care for right now
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -186,7 +187,11 @@ DEF_OP(Syscall) {
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -302,6 +307,42 @@ DEF_OP(CPUID) {
|
||||
mov(Dst.second, rdx);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
|
||||
// CPUID ABI
|
||||
// this: rdi
|
||||
// Function: rsi
|
||||
//
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
mov(Dst.first.cvt32(), eax);
|
||||
mov(Dst.second, rax);
|
||||
shr(Dst.second, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -314,6 +355,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
+27
-4
@@ -308,16 +308,13 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto SrcRAX = GetSrc<RA_64>(Op->RAX.ID());
|
||||
const auto SrcRDX = GetSrc<RA_64>(Op->RDX.ID());
|
||||
|
||||
// Encode the size check into the 8th bit to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(rdi, SrcRAX);
|
||||
mov(rsi, SrcRDX);
|
||||
|
||||
@@ -444,6 +441,11 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
@@ -815,12 +817,33 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
auto JITBlockTail = getCurr<JITCodeTail*>();
|
||||
setSize(getSize() + sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = getCurr<uint8_t *>();
|
||||
auto JITRIPEntries = getCurr<JITRIPReconstructEntries*>();
|
||||
|
||||
setSize(getSize() + sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = getCurr<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
@@ -313,6 +313,7 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
|
||||
+6
-1
@@ -22,6 +22,9 @@ namespace FEXCore::CodeSerialize {
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
|
||||
@@ -48,13 +51,14 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
unsigned _Pad : 17;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
@@ -71,6 +75,7 @@ namespace FEXCore::CodeSerialize {
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
|
||||
+84
-33
@@ -35,6 +35,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
constexpr size_t SyscallArgs = 7;
|
||||
using SyscallArray = std::array<uint64_t, SyscallArgs>;
|
||||
|
||||
size_t NumArguments{};
|
||||
const SyscallArray *GPRIndexes {};
|
||||
static constexpr SyscallArray GPRIndexes_64 = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
@@ -54,13 +55,26 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
FEXCore::X86State::REG_RDI,
|
||||
FEXCore::X86State::REG_RBP,
|
||||
};
|
||||
static_assert(GPRIndexes_64.size() == GPRIndexes_32.size());
|
||||
|
||||
static std::array<uint64_t, SyscallArgs> GPRIndexes_Hangover = {
|
||||
static constexpr SyscallArray GPRIndexes_Hangover = {
|
||||
FEXCore::X86State::REG_RCX,
|
||||
};
|
||||
|
||||
size_t NumArguments{};
|
||||
static constexpr SyscallArray GPRIndexes_Win64 = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_R10,
|
||||
FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_R8,
|
||||
FEXCore::X86State::REG_R9,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
};
|
||||
|
||||
static constexpr SyscallArray GPRIndexes_Win32 = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
};
|
||||
|
||||
SyscallFlags DefaultSyscallFlags = FEXCore::IR::SyscallFlags::DEFAULT;
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64) {
|
||||
@@ -68,9 +82,19 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
GPRIndexes = &GPRIndexes_64;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX32) {
|
||||
NumArguments = GPRIndexes_64.size();
|
||||
NumArguments = GPRIndexes_32.size();
|
||||
GPRIndexes = &GPRIndexes_32;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_WIN64) {
|
||||
NumArguments = 6;
|
||||
GPRIndexes = &GPRIndexes_Win64;
|
||||
DefaultSyscallFlags = FEXCore::IR::SyscallFlags::NORETURNEDRESULT;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_WIN32) {
|
||||
NumArguments = 2;
|
||||
GPRIndexes = &GPRIndexes_Win32;
|
||||
DefaultSyscallFlags = FEXCore::IR::SyscallFlags::NORETURNEDRESULT;
|
||||
}
|
||||
else if (OSABI == FEXCore::HLE::SyscallOSABI::OS_HANGOVER) {
|
||||
NumArguments = 1;
|
||||
GPRIndexes = &GPRIndexes_Hangover;
|
||||
@@ -109,13 +133,20 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
Arguments[4],
|
||||
Arguments[5],
|
||||
Arguments[6],
|
||||
FEXCore::IR::SyscallFlags::DEFAULT);
|
||||
DefaultSyscallFlags);
|
||||
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_HANGOVER) {
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_HANGOVER &&
|
||||
(DefaultSyscallFlags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Hangover doesn't want us returning a result here
|
||||
// syscall is being abused as a thunk for now.
|
||||
StoreGPRRegister(X86State::REG_RAX, SyscallOp);
|
||||
}
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
_ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
@@ -1513,7 +1544,7 @@ void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadGPRRegister(X86State::REG_RAX, 1, 8);
|
||||
|
||||
// Clear bits that aren't supposed to be set
|
||||
Src = _And(Src, _Constant(~0b101000));
|
||||
Src = _Andn(Src, _Constant(0b101000));
|
||||
|
||||
// Set the bit that is always set here
|
||||
Src = _Or(Src, _Constant(0b10));
|
||||
@@ -1748,6 +1779,18 @@ void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, _Bfe(32, 0, Result_Upper));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XGetBVOp(OpcodeArgs) {
|
||||
OrderedNode *Function = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
auto Res = _XGetBV(Function);
|
||||
|
||||
OrderedNode *Result_Lower = _ExtractElementPair(Res, 0);
|
||||
OrderedNode *Result_Upper = _ExtractElementPair(Res, 1);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, Result_Lower);
|
||||
StoreGPRRegister(X86State::REG_RDX, Result_Upper);
|
||||
}
|
||||
|
||||
template<bool SHL1Bit>
|
||||
void OpDispatchBuilder::SHLOp(OpcodeArgs) {
|
||||
OrderedNode *Src{};
|
||||
@@ -3807,7 +3850,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
OrderedNode *Counter = LoadGPRRegister(X86State::REG_RCX);
|
||||
auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC);
|
||||
|
||||
auto Result = _MemSet(CTX->IsTSOEnabled(), Size, Segment ?: InvalidNode, Dest, Src, Counter, DF);
|
||||
auto Result = _MemSet(CTX->IsAtomicTSOEnabled(), Size, Segment ?: InvalidNode, Dest, Src, Counter, DF);
|
||||
StoreGPRRegister(X86State::REG_RCX, _Constant(0));
|
||||
StoreGPRRegister(X86State::REG_RDI, Result);
|
||||
}
|
||||
@@ -3834,7 +3877,7 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
auto DstSegment = GetSegment(0, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
auto SrcSegment = GetSegment(Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
auto Result = _MemCpy(CTX->IsTSOEnabled(), Size,
|
||||
auto Result = _MemCpy(CTX->IsAtomicTSOEnabled(), Size,
|
||||
DstSegment ?: InvalidNode,
|
||||
SrcSegment ?: InvalidNode,
|
||||
DstAddr, SrcAddr, Counter, DF);
|
||||
@@ -5065,8 +5108,8 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
// TODO: Fix the instructions doing partial writes rather than dealing with it here.
|
||||
auto SrcVector = LoadXMMRegister(gprIndex);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Class != IR::GPRClass, "Partial writes from GPR not allowed. Instruction: {}",
|
||||
Op->TableInfo->Name);
|
||||
LOGMAN_THROW_A_FMT(Class != IR::GPRClass, "Partial writes from GPR not allowed. Instruction: {}",
|
||||
Op->TableInfo->Name);
|
||||
|
||||
// OpSize of 16 is special in that it is expected to zero the upper bits of the 256-bit operation.
|
||||
// TODO: Longer term we should enforce the difference between zero and insert.
|
||||
@@ -5309,7 +5352,6 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
ALUOpImpl(Op, ALUIROp, AtomicFetchOp, RequiresMask);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
IR::BreakDefinition Reason;
|
||||
bool SetRIPToNext = false;
|
||||
@@ -5318,14 +5360,19 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
case 0xCD: { // INT imm8
|
||||
uint8_t Literal = Op->Src[0].Data.Literal.Value;
|
||||
|
||||
if (Literal == 0x80) {
|
||||
#ifndef _WIN32
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x80;
|
||||
#else
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x2E;
|
||||
#endif
|
||||
if (Literal == SYSCALL_LITERAL) {
|
||||
// Syscall on linux
|
||||
SyscallOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Reason.ErrorRegister = Literal << 3 | (0b010);
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
// GP is raised when task-gate isn't setup to be valid
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
@@ -5333,33 +5380,33 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
}
|
||||
case 0xCE: // INTO
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_OF;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
case 0xF1: // INT1
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.Signal = Core::FAULT_SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_DB;
|
||||
Reason.si_code = 1;
|
||||
SetRIPToNext = true;
|
||||
break;
|
||||
case 0xF4: { // HLT
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
}
|
||||
case 0x0B: // UD2
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGILL;
|
||||
Reason.Signal = Core::FAULT_SIGILL;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_UD;
|
||||
Reason.si_code = 2;
|
||||
break;
|
||||
case 0xCC: // INT3
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.Signal = Core::FAULT_SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_BP;
|
||||
Reason.si_code = 0x80;
|
||||
SetRIPToNext = true;
|
||||
@@ -5406,11 +5453,6 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
_Break(Reason);
|
||||
}
|
||||
}
|
||||
#else
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
ERROR_AND_DIE_FMT("Unknown INTOp instruction?");
|
||||
}
|
||||
#endif
|
||||
|
||||
void OpDispatchBuilder::TZCNT(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
@@ -5435,11 +5477,6 @@ void OpDispatchBuilder::MOVBEOp(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void OpDispatchBuilder::FenceOp(OpcodeArgs) {
|
||||
_Fence({FenceType});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CLWB(OpcodeArgs) {
|
||||
OrderedNode *DestMem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
DestMem = AppendSegmentOffset(DestMem, Op->Flags);
|
||||
@@ -5452,6 +5489,15 @@ void OpDispatchBuilder::CLFLUSHOPT(OpcodeArgs) {
|
||||
_CacheLineClear(DestMem, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LoadFenceOrXRSTOR(OpcodeArgs) {
|
||||
// 0xE8 signifies LFENCE
|
||||
if (Op->ModRM == 0xE8) {
|
||||
_Fence(IR::Fence_Load);
|
||||
} else {
|
||||
XRstorOpImpl(Op);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MemFenceOrXSAVEOPT(OpcodeArgs) {
|
||||
if (Op->ModRM == 0xF0) {
|
||||
// 0xF0 is MFENCE
|
||||
@@ -5992,7 +6038,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(3, 0b01, 0x4B), 1, &OpDispatchBuilder::AVXVectorVariableBlend<8>},
|
||||
{OPD(3, 0b01, 0x4C), 1, &OpDispatchBuilder::AVXVectorVariableBlend<1>},
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::VAESKeyGenAssistOp},
|
||||
@@ -6669,9 +6717,10 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 1), 1, &OpDispatchBuilder::FXRStoreOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 2), 1, &OpDispatchBuilder::LDMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 3), 1, &OpDispatchBuilder::STMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::FenceOp<FEXCore::IR::Fence_Load.Val>}, //LFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, //MFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, //SFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 4), 1, &OpDispatchBuilder::XSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::LoadFenceOrXRSTOR}, // LFENCE (or XRSTOR)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, // MFENCE (or XSAVEOPT)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, // SFENCE (or CLFLUSH)
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
@@ -6704,7 +6753,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> SecondaryModRMExtensionOpTable[] = {
|
||||
// REG /2
|
||||
{((1 << 3) | 0), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{((1 << 3) | 0), 1, &OpDispatchBuilder::XGetBVOp},
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
@@ -7283,7 +7332,9 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
|
||||
+51
-6
@@ -149,6 +149,31 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool CanHaveSideEffects(FEXCore::X86Tables::X86InstInfo const* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
|
||||
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
}
|
||||
|
||||
auto CanHaveSideEffects = false;
|
||||
|
||||
auto HasPotentialMemoryAccess = [](X86Tables::DecodedOperand const &Operand) -> bool {
|
||||
if (Operand.IsNone()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// This isn't guaranteed that all of these types will access memory, but be safe.
|
||||
return Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB();
|
||||
};
|
||||
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Dest);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[0]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[1]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[2]);
|
||||
return CanHaveSideEffects;
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
|
||||
@@ -219,6 +244,7 @@ public:
|
||||
void MOVOffsetOp(OpcodeArgs);
|
||||
void CMOVOp(OpcodeArgs);
|
||||
void CPUIDOp(OpcodeArgs);
|
||||
void XGetBVOp(OpcodeArgs);
|
||||
template<bool SHL1Bit>
|
||||
void SHLOp(OpcodeArgs);
|
||||
void SHLImmediateOp(OpcodeArgs);
|
||||
@@ -491,7 +517,9 @@ public:
|
||||
void VPALIGNROp(OpcodeArgs);
|
||||
|
||||
void VPCMPESTRIOp(OpcodeArgs);
|
||||
void VPCMPESTRMOp(OpcodeArgs);
|
||||
void VPCMPISTRIOp(OpcodeArgs);
|
||||
void VPCMPISTRMOp(OpcodeArgs);
|
||||
|
||||
void VPERM2Op(OpcodeArgs);
|
||||
void VPERMDOp(OpcodeArgs);
|
||||
@@ -696,6 +724,8 @@ public:
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
void XSaveOp(OpcodeArgs);
|
||||
|
||||
void PAlignrOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void UCOMISxOp(OpcodeArgs);
|
||||
@@ -745,11 +775,9 @@ public:
|
||||
void PHADDS(OpcodeArgs);
|
||||
void PHSUBS(OpcodeArgs);
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void FenceOp(OpcodeArgs);
|
||||
|
||||
void CLWB(OpcodeArgs);
|
||||
void CLFLUSHOPT(OpcodeArgs);
|
||||
void LoadFenceOrXRSTOR(OpcodeArgs);
|
||||
void MemFenceOrXSAVEOPT(OpcodeArgs);
|
||||
void StoreFenceOrCLFlush(OpcodeArgs);
|
||||
void CLZeroOp(OpcodeArgs);
|
||||
@@ -862,7 +890,7 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit);
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
OrderedNode* PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2);
|
||||
@@ -950,6 +978,23 @@ private:
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
|
||||
void XSaveOpImpl(OpcodeArgs);
|
||||
void SaveX87State(OpcodeArgs, OrderedNode *MemBase);
|
||||
void SaveSSEState(OrderedNode *MemBase);
|
||||
void SaveMXCSRState(OrderedNode *MemBase);
|
||||
void SaveAVXState(OrderedNode *MemBase);
|
||||
|
||||
void XRstorOpImpl(OpcodeArgs);
|
||||
void RestoreX87State(OrderedNode *MemBase);
|
||||
void RestoreSSEState(OrderedNode *MemBase);
|
||||
void RestoreMXCSRState(OrderedNode *MXCSR);
|
||||
void RestoreAVXState(OrderedNode *MemBase);
|
||||
void DefaultX87State(OpcodeArgs);
|
||||
void DefaultSSEState();
|
||||
void DefaultAVXState();
|
||||
|
||||
OrderedNode *GetMXCSR();
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
@@ -1555,14 +1600,14 @@ private:
|
||||
uint64_t Entry;
|
||||
|
||||
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
if (CTX->IsTSOEnabled())
|
||||
if (CTX->IsAtomicTSOEnabled())
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
if (CTX->IsTSOEnabled())
|
||||
if (CTX->IsAtomicTSOEnabled())
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
@@ -52,10 +52,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
auto Tmp = _And(_Lshr(Src, _Constant(FlagOffset)), OneConst);
|
||||
auto Tmp = _Bfe(4, 1, FlagOffset, Src);
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
}
|
||||
}
|
||||
@@ -304,28 +303,11 @@ void OpDispatchBuilder::CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, O
|
||||
// OF
|
||||
// Signed
|
||||
{
|
||||
auto NegOne = _Constant(~0ULL);
|
||||
auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne);
|
||||
auto XorOp1 = _Not(_Xor(Src1, Src2));
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (Size) {
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
case 16:
|
||||
AndOp1 = _Bfe(1, 15, AndOp1);
|
||||
break;
|
||||
case 32:
|
||||
AndOp1 = _Bfe(1, 31, AndOp1);
|
||||
break;
|
||||
case 64:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", Size);
|
||||
break;
|
||||
}
|
||||
AndOp1 = _Bfe(1, Size - 1, AndOp1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
@@ -376,24 +358,7 @@ void OpDispatchBuilder::CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, O
|
||||
auto XorOp1 = _Xor(Src1, Src2);
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (SrcSize) {
|
||||
case 1:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
case 2:
|
||||
AndOp1 = _Bfe(1, 15, AndOp1);
|
||||
break;
|
||||
case 4:
|
||||
AndOp1 = _Bfe(1, 31, AndOp1);
|
||||
break;
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", SrcSize);
|
||||
break;
|
||||
}
|
||||
AndOp1 = _Bfe(1, SrcSize * 8 - 1, AndOp1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
@@ -492,29 +457,12 @@ void OpDispatchBuilder::CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, O
|
||||
|
||||
// OF
|
||||
{
|
||||
auto NegOne = _Constant(~0ULL);
|
||||
auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne);
|
||||
auto XorOp1 = _Not(_Xor(Src1, Src2));
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
|
||||
OrderedNode *AndOp1 = _And(XorOp1, XorOp2);
|
||||
|
||||
switch (SrcSize) {
|
||||
case 1:
|
||||
AndOp1 = _Bfe(1, 7, AndOp1);
|
||||
break;
|
||||
case 2:
|
||||
AndOp1 = _Bfe(1, 15, AndOp1);
|
||||
break;
|
||||
case 4:
|
||||
AndOp1 = _Bfe(1, 31, AndOp1);
|
||||
break;
|
||||
case 8:
|
||||
AndOp1 = _Bfe(1, 63, AndOp1);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown BFE size: {}", SrcSize);
|
||||
break;
|
||||
}
|
||||
AndOp1 = _Bfe(1, SrcSize * 8 - 1, AndOp1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
}
|
||||
|
||||
+303
-40
@@ -2455,6 +2455,76 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
SaveX87State(Op, Mem);
|
||||
SaveSSEState(Mem);
|
||||
SaveMXCSRState(Mem);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOp(OpcodeArgs) {
|
||||
XSaveOpImpl(Op);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// NOTE: Mask should be EAX and EDX concatenated, but we only need to test
|
||||
// for features that are in the lower 32 bits, so EAX only is sufficient.
|
||||
OrderedNode *Mask = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *Base = XSaveBase();
|
||||
|
||||
const auto StoreIfFlagSet = [&](uint32_t BitIndex, auto fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
fn();
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetFalseJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
StoreIfFlagSet(0, [this, Op, Base] { SaveX87State(Op, Base); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveSSEState(Base); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
StoreIfFlagSet(2, [this, Base] { SaveAVXState(Base); });
|
||||
}
|
||||
|
||||
// We need to save MXCSR and MXCSR_MASK if either SSE or AVX are requested to be saved
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveMXCSRState(Base); }, 2);
|
||||
}
|
||||
|
||||
// Update XSTATE_BV region of the XSAVE header
|
||||
{
|
||||
OrderedNode *HeaderOffset = _Add(Base, _Constant(512));
|
||||
|
||||
// NOTE: We currently only support the first 3 bits (x87, SSE, and AVX)
|
||||
OrderedNode *RequestedFeatures = _Bfe(3, 0, Mask);
|
||||
|
||||
// XSTATE_BV section of the header is 8 bytes in size, but we only really
|
||||
// care about setting at most 3 bits in the first byte. We zero out the rest.
|
||||
_StoreMem(GPRClass, 8, HeaderOffset, RequestedFeatures);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveX87State(OpcodeArgs, OrderedNode *MemBase) {
|
||||
// Saves 512bytes to the memory location provided
|
||||
// Header changes depending on if REX.W is set or not
|
||||
if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) {
|
||||
@@ -2472,12 +2542,12 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, 2, Mem, FCW, 2);
|
||||
_StoreMem(GPRClass, 2, MemBase, FCW, 2);
|
||||
}
|
||||
|
||||
{
|
||||
// We must construct the FSW from our various bits
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
OrderedNode *FSW = _Constant(0);
|
||||
auto Top = GetX87Top();
|
||||
FSW = _Or(FSW, _Lshl(Top, _Constant(11)));
|
||||
@@ -2496,7 +2566,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
_StoreMem(GPRClass, 2, MemLocation, FTW, 2);
|
||||
}
|
||||
@@ -2545,33 +2615,142 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, MMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(OrderedNode *MemBase) {
|
||||
OrderedNode *MXCSR = GetMXCSR();
|
||||
OrderedNode *MXCSRLocation = _Add(MemBase, _Constant(24));
|
||||
_StoreMem(GPRClass, 4, MXCSRLocation, MXCSR, 4);
|
||||
|
||||
// Store the mask for all bits.
|
||||
OrderedNode *MXCSRMaskLocation = _Add(MXCSRLocation, _Constant(4));
|
||||
_StoreMem(GPRClass, 4, MXCSRMaskLocation, _Constant(0xFFFF), 4);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *Upper = _VDupElement(32, 16, LoadXMMRegister(i), 1);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, Upper, 16);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetMXCSR() {
|
||||
// Default MXCSR Value
|
||||
OrderedNode *MXCSR = _Constant(0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
return _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
RestoreX87State(Mem);
|
||||
RestoreSSEState(Mem);
|
||||
|
||||
OrderedNode *MXCSRLocation = _Add(Mem, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// Set up base address for the XSAVE region to restore from, and also read the
|
||||
// XSTATE_BV bit flags out of the XSTATE header.
|
||||
OrderedNode *Base = XSaveBase();
|
||||
OrderedNode *Mask = _LoadMem(GPRClass, 8, _Add(Base, _Constant(512)), 8);
|
||||
|
||||
// If a bit in our XSTATE_BV is set, then we restore from that region of the XSAVE area,
|
||||
// otherwise, if not set, then we need to set the relevant data the bit corresponds to
|
||||
// to it's defined initial configuration.
|
||||
const auto RestoreIfFlagSetOrDefault = [&](uint32_t BitIndex, auto restore_fn, auto default_fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto RestoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, RestoreBlock);
|
||||
SetCurrentCodeBlock(RestoreBlock);
|
||||
{
|
||||
restore_fn();
|
||||
}
|
||||
auto RestoreExitJump = _Jump();
|
||||
auto DefaultBlock = CreateNewCodeBlockAfter(RestoreBlock);
|
||||
auto ExitBlock = CreateNewCodeBlockAfter(DefaultBlock);
|
||||
SetJumpTarget(RestoreExitJump, ExitBlock);
|
||||
SetFalseJumpTarget(CondJump, DefaultBlock);
|
||||
SetCurrentCodeBlock(DefaultBlock);
|
||||
{
|
||||
default_fn();
|
||||
}
|
||||
auto DefaultExitJump = _Jump();
|
||||
SetJumpTarget(DefaultExitJump, ExitBlock);
|
||||
SetCurrentCodeBlock(ExitBlock);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(0,
|
||||
[this, Base] { RestoreX87State(Base); },
|
||||
[this, Op] { DefaultX87State(Op); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] { RestoreSSEState(Base); },
|
||||
[this] { DefaultSSEState(); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(2,
|
||||
[this, Base] { RestoreAVXState(Base); },
|
||||
[this] { DefaultAVXState(); });
|
||||
}
|
||||
|
||||
{
|
||||
// We need to restore the MXCSR if either SSE or AVX are requested to be saved
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] {
|
||||
OrderedNode *MXCSRLocation = _Add(Base, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
},
|
||||
[] { /* Intentionally do nothing*/ }, 2);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreX87State(OrderedNode *MemBase) {
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, MemBase, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
{
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
|
||||
// Strip out the FSW information
|
||||
@@ -2591,26 +2770,78 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto NewFTW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
}
|
||||
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreMXCSRState(OrderedNode *MXCSR) {
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, MXCSR);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
OrderedNode *YMMHReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
OrderedNode *YMM = _VInsElement(32, 16, 1, 0, XMMReg, YMMHReg);
|
||||
StoreXMMRegister(i, YMM);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
|
||||
// We can piggy-back on FNINIT's implementation, since
|
||||
// it performs the same behavior as required by XRSTOR for resetting flags
|
||||
FNINIT(Op);
|
||||
|
||||
// On top of resetting the flags to a default state, we also need to clear
|
||||
// all of the ST0-7/MM0-7 registers to zero.
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::MM_REG_SIZE);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
_StoreContext(16, FPRClass, ZeroVector, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultSSEState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
StoreXMMRegister(i, ZeroVector);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultAVXState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
OrderedNode* Reg = LoadXMMRegister(i);
|
||||
OrderedNode* Dst = _VMov(16, Reg);
|
||||
StoreXMMRegister(i, Dst);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
@@ -2676,18 +2907,11 @@ void OpDispatchBuilder::UCOMISxOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::LDMXCSR(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, Dest);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
RestoreMXCSRState(Dest);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::STMXCSR(OpcodeArgs) {
|
||||
// Default MXCSR
|
||||
OrderedNode *MXCSR = _Constant(32, 0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
MXCSR = _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
|
||||
StoreResult(GPRClass, Op, MXCSR, -1);
|
||||
StoreResult(GPRClass, Op, GetMXCSR(), -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PACKUSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
@@ -4591,9 +4815,9 @@ void OpDispatchBuilder::VPERMILRegOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPERMILRegOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit) {
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[1] needs to be a literal");
|
||||
const auto Control = Op->Src[1].Data.Literal.Value;
|
||||
const uint16_t Control = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
// SSE4.2 string instructions modify flags, so invalidate
|
||||
// any previously deferred flags.
|
||||
@@ -4611,32 +4835,68 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit) {
|
||||
OrderedNode *IntermediateResult{};
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
//
|
||||
// While the control bit immediate for the instruction itself is only ever 8 bits
|
||||
// in size, we use it as a 16-bit value so that we can use the 8th bit to signify
|
||||
// whether or not RAX and RDX should be interpreted as a 64-bit value.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is64Bit = SrcSize == 8;
|
||||
const auto NewControl = uint16_t(Control | (uint16_t(Is64Bit) << 8));
|
||||
|
||||
OrderedNode *SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(SrcSize, Src1, Src2, SrcRAX, SrcRDX, Control);
|
||||
IntermediateResult = _VPCMPESTRX(Src1, Src2, SrcRAX, SrcRDX, NewControl);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
OrderedNode *ResultNoFlags = _And(IntermediateResult, _Constant(0xFFFF));
|
||||
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
OrderedNode *ZeroConst = _Constant(0);
|
||||
|
||||
const auto ECXResult = [&]() -> OrderedNode* {
|
||||
if (IsMask) {
|
||||
// For the masked variant of the instructions, if control[6] is set, then we
|
||||
// need to expand the intermediate result into a byte or word mask (depending
|
||||
// on data size specified in control[1]) along the entire length of XMM0,
|
||||
// where set bits in the intermediate result set the corresponding entry
|
||||
// in XMM0 to all 1s and unset bits set the corresponding entry to all 0s.
|
||||
//
|
||||
// If control[6] is not set, then we just store the intermediate result as-is
|
||||
// into the least significant bits of XMM0 and zero extend it.
|
||||
const auto IsExpandedMask = (Control & 0b0100'0000) != 0;
|
||||
|
||||
if (IsExpandedMask) {
|
||||
// We need to iterate over the intermediate result and
|
||||
// expand the mask into XMM0 elements.
|
||||
const auto ElementSize = 1U << (Control & 1);
|
||||
const auto NumElements = 16U >> (Control & 1);
|
||||
|
||||
OrderedNode *Result = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
OrderedNode *SignBit = _Sbfe(1, i, IntermediateResult);
|
||||
Result = _VInsGPR(Core::CPUState::XMM_SSE_REG_SIZE, ElementSize, i, Result, SignBit);
|
||||
}
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
// We insert the intermediate result as-is.
|
||||
StoreXMMRegister(0, _VCastFromGPR(16, 2, IntermediateResult));
|
||||
}
|
||||
} else {
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
const auto UseMSBIndex = (Control & 0b0100'0000) != 0;
|
||||
|
||||
OrderedNode *ResultNoFlags = _Bfe(16, 0, IntermediateResult);
|
||||
|
||||
OrderedNode *IfZero = _Constant(16 >> (Control & 1));
|
||||
OrderedNode *IfNotZero = UseMSBIndex ? _FindMSB(ResultNoFlags)
|
||||
: _FindLSB(ResultNoFlags);
|
||||
|
||||
return _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
}();
|
||||
OrderedNode *Result = _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RCX, Result, 4);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags.
|
||||
// We use the top 16-bits of the result to store the flags
|
||||
@@ -4656,16 +4916,19 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit) {
|
||||
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(ZeroConst);
|
||||
|
||||
// ... and we're done!
|
||||
StoreGPRRegister(X86State::REG_RCX, ECXResult, 4);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCMPESTRIOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true);
|
||||
PCMPXSTRXOpImpl(Op, true, false);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPESTRMOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true, true);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPISTRIOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false);
|
||||
PCMPXSTRXOpImpl(Op, false, false);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPISTRMOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false, true);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1374,8 +1374,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
|
||||
SrcCond = _Sbfe(1, 0, SrcCond);
|
||||
|
||||
OrderedNode *VecCond = _VCastFromGPR(16, 8, SrcCond);
|
||||
VecCond = _VInsGPR(16, 8, 1, VecCond, SrcCond);
|
||||
OrderedNode *VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
|
||||
@@ -146,10 +146,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
@@ -169,7 +169,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_DEBUG , 1, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
|
||||
@@ -43,9 +43,9 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
@@ -334,7 +334,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 1), 1, X86InstInfo{"FXRSTOR", TYPE_INST, FLAGS_MODRM, 0, nullptr}}, // MMX/x87
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, X86InstInfo{"LDMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, X86InstInfo{"STMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 5), 1, X86InstInfo{"LFENCE/XRSTOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 6), 1, X86InstInfo{"MFENCE/XSAVEOPT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 7), 1, X86InstInfo{"SFENCE/CLFLUSH", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
@@ -22,7 +22,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x02, 1, X86InstInfo{"LAR", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x07, 1, X86InstInfo{"SYSRET", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
@@ -456,9 +456,9 @@ void InitializeVEXTables() {
|
||||
{OPD(3, 0b01, 0x5E), 1, X86InstInfo{"VMFSUBADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x5F), 1, X86InstInfo{"VFMSUBADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, X86InstInfo{"VPCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x60), 1, X86InstInfo{"VPCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x61), 1, X86InstInfo{"VPCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x62), 1, X86InstInfo{"VPCMPISTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x62), 1, X86InstInfo{"VPCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x63), 1, X86InstInfo{"VPCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x68), 1, X86InstInfo{"VFMADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -361,6 +361,13 @@ constexpr InstFlagType SIZE_128BIT = 0b101;
|
||||
constexpr InstFlagType SIZE_256BIT = 0b110;
|
||||
constexpr InstFlagType SIZE_64BITDEF = 0b111; // Default mode is 64bit instead of typical 32bit
|
||||
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY;
|
||||
#else
|
||||
// Syscall ends a block on WIN32 because the instruction can update the CPU's RIP.
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY | FLAGS_BLOCK_END;
|
||||
#endif
|
||||
|
||||
constexpr InstFlagType GetSizeDstFlags(InstFlagType Flags) { return (Flags >> FLAGS_SIZE_DST_OFF) & SIZE_MASK; }
|
||||
constexpr InstFlagType GetSizeSrcFlags(InstFlagType Flags) { return (Flags >> FLAGS_SIZE_SRC_OFF) & SIZE_MASK; }
|
||||
|
||||
|
||||
+2
-1
@@ -53,8 +53,9 @@ static __attribute__((aligned(16), naked, section("HostToGuestTrampolineTemplate
|
||||
);
|
||||
#elif defined(_M_ARM_64)
|
||||
asm(
|
||||
// x11 is part of the custom ABI and needs to point to the TrampolineInstanceInfo.
|
||||
"ldr x16, 0f \n"
|
||||
"adr x11, 0f \n"
|
||||
"ldr x16, [x11] \n"
|
||||
"br x16 \n"
|
||||
// Manually align to the next 8-byte boundary
|
||||
// NOTE: GCC over-aligns to a full page when using .align directives on ARM (last tested on GCC 11.2)
|
||||
|
||||
+3
-3
@@ -251,8 +251,8 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
PreGenerateIRFetchResult Result{};
|
||||
|
||||
@@ -306,7 +306,7 @@ namespace FEXCore::IR {
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
|
||||
+1
-1
@@ -105,7 +105,7 @@ namespace FEXCore::IR {
|
||||
uint64_t Length {};
|
||||
bool GeneratedIR {};
|
||||
};
|
||||
[[nodiscard]] PreGenerateIRFetchResult PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList);
|
||||
[[nodiscard]] PreGenerateIRFetchResult PreGenerateIRFetch(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, FEXCore::IR::IRListView *IRList);
|
||||
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState *Thread,
|
||||
void* CodePtr,
|
||||
|
||||
+10
-4
@@ -297,6 +297,13 @@
|
||||
],
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
"GPRPair = XGetBV GPR:$Function": {
|
||||
"Desc": ["Calls in to the XCR handler function to return emulated XCR",
|
||||
"Returns a 64bit GPR pair that fits emulated EAX, EDX respectively"
|
||||
],
|
||||
"DestSize": "8",
|
||||
"NumElements": "2"
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
@@ -714,7 +721,8 @@
|
||||
"Desc": ["Integer binary not",
|
||||
"op:",
|
||||
"Dest = ~Src"
|
||||
]
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Popcount GPR:$Src": {
|
||||
"Desc": ["Population count of source register",
|
||||
@@ -1397,7 +1405,7 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"GPR = VPCMPESTRX u8:$GPRSize, FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u8:$Control": {
|
||||
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
@@ -1406,7 +1414,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
@@ -1418,7 +1425,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
|
||||
+1
-1
@@ -103,7 +103,7 @@ static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNode
|
||||
if (ArgID.IsInvalid()) {
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%ssa" << ArgID;
|
||||
*out << "%ssa" << std::dec << ArgID;
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
|
||||
@@ -314,6 +314,36 @@ namespace {
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// _pad2
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalRefCount
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalRefCount),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalFaultAddress
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalFaultAddress),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
|
||||
[[maybe_unused]] size_t ClassifiedStructSize{};
|
||||
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
|
||||
@@ -393,6 +423,10 @@ namespace {
|
||||
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
|
||||
+12
@@ -340,5 +340,17 @@ namespace FEXCore::Allocator {
|
||||
::munmap(Region.Ptr, Region.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Alloc64) {
|
||||
Alloc64->LockBeforeFork(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {
|
||||
if (Alloc64) {
|
||||
Alloc64->UnlockAfterFork(Thread, Child);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child);
|
||||
}
|
||||
+28
-5
@@ -5,7 +5,7 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXHeaderUtils/ScopedSignalMask.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -29,6 +29,16 @@
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
|
||||
thread_local FEXCore::Core::InternalThreadState *TLSThread{};
|
||||
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
TLSThread = Thread;
|
||||
}
|
||||
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
TLSThread = nullptr;
|
||||
}
|
||||
|
||||
class OSAllocator_64Bit final : public Alloc::HostAllocator {
|
||||
public:
|
||||
OSAllocator_64Bit();
|
||||
@@ -39,6 +49,19 @@ namespace Alloc::OSAllocator {
|
||||
void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) override;
|
||||
int Munmap(void *addr, size_t length) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override {
|
||||
AllocationMutex.lock();
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override {
|
||||
if (Child) {
|
||||
AllocationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
AllocationMutex.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Upper bound is the maximum virtual address space of the host processor
|
||||
uintptr_t UPPER_BOUND = (1ULL << 57);
|
||||
@@ -129,7 +152,7 @@ namespace Alloc::OSAllocator {
|
||||
LiveRegionListType *LiveRegions{};
|
||||
|
||||
Alloc::ForwardOnlyIntrusiveArenaAllocator *ObjectAlloc{};
|
||||
std::mutex AllocationMutex{};
|
||||
FEXCore::ForkableUniqueMutex AllocationMutex;
|
||||
void DetermineVASize();
|
||||
|
||||
LiveVMARegion *MakeRegionActive(ReservedRegionListType::iterator ReservedIterator, uint64_t UsedSize) {
|
||||
@@ -248,7 +271,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -436,7 +459,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -561,7 +584,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
@@ -6,6 +6,10 @@
|
||||
#include <cstdint>
|
||||
#include <sys/types.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace Alloc {
|
||||
// HostAllocator is just a page pased slab allocator
|
||||
// Similar to mmap and munmap only mapping at the page level
|
||||
@@ -18,6 +22,9 @@ namespace Alloc {
|
||||
|
||||
virtual void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) { return nullptr; }
|
||||
virtual int Munmap(void *addr, size_t length) { return -1; }
|
||||
|
||||
virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
};
|
||||
|
||||
class GlobalAllocator {
|
||||
@@ -36,5 +43,7 @@ namespace Alloc {
|
||||
}
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
}
|
||||
+3
-3
@@ -2043,8 +2043,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
@@ -2068,7 +2068,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
return std::make_pair(true, -4);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
|
||||
@@ -22,6 +22,11 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
|
||||
std::pair<bool, int32_t> HandleUnalignedAccess(bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t *GPRs) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
+9
-9
@@ -1,4 +1,5 @@
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -22,6 +23,7 @@ namespace FEXCore::Telemetry {
|
||||
"32bit CAS Tear",
|
||||
"64bit CAS Tear",
|
||||
"128bit CAS Tear",
|
||||
"Crash mask",
|
||||
};
|
||||
void Initialize() {
|
||||
auto DataDirectory = Config::GetDataDirectory();
|
||||
@@ -35,7 +37,6 @@ namespace FEXCore::Telemetry {
|
||||
}
|
||||
|
||||
void Shutdown(fextl::string const &ApplicationName) {
|
||||
#ifndef _WIN32
|
||||
auto DataDirectory = Config::GetDataDirectory();
|
||||
DataDirectory += "Telemetry/" + ApplicationName + ".telem";
|
||||
|
||||
@@ -45,20 +46,19 @@ namespace FEXCore::Telemetry {
|
||||
FHU::Filesystem::CopyFile(DataDirectory, Backup, FHU::Filesystem::CopyOptions::OVERWRITE_EXISTING);
|
||||
}
|
||||
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
int fd = open(DataDirectory.c_str(), O_CREAT | O_WRONLY | O_TRUNC | O_CLOEXEC, USER_PERMS);
|
||||
auto File = FEXCore::File::File(DataDirectory.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
|
||||
if (fd != -1) {
|
||||
if (File.IsValid()) {
|
||||
for (size_t i = 0; i < TelemetryType::TYPE_LAST; ++i) {
|
||||
auto &Name = TelemetryNames.at(i);
|
||||
auto &Data = TelemetryValues.at(i);
|
||||
auto Output = fextl::fmt::format("{}: {}\n", Name, *Data);
|
||||
write(fd, Output.c_str(), Output.size());
|
||||
fextl::fmt::print(File, "{}: {}\n", Name, *Data);
|
||||
}
|
||||
fsync(fd);
|
||||
close(fd);
|
||||
File.Flush();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Value &GetObject(TelemetryType Type) {
|
||||
|
||||
@@ -244,50 +244,4 @@ namespace Type {
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
};
|
||||
|
||||
// Application loaders
|
||||
class FEX_DEFAULT_VISIBILITY OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit OptionMapper(FEXCore::Config::LayerType Layer);
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char *ConfigName, const char *ConfigString);
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
|
||||
/**
|
||||
* @brief Loads the main application config
|
||||
*
|
||||
* @param File Optional override to load a specific config file in to the main layer
|
||||
* Shouldn't be commonly used
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
*
|
||||
* @param Filename Application filename component
|
||||
* @param Global Load the global configuration or user accessible file
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
*
|
||||
* @param _envp[] The environment array from main
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]);
|
||||
|
||||
}
|
||||
@@ -89,6 +89,28 @@ namespace CPU {
|
||||
size_t Size;
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
// Offset after this block to the start of the RIP entries.
|
||||
uint32_t OffsetToRIPEntries;
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using 16-bit entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
struct JITRIPReconstructEntries {
|
||||
// The Host PC offset from the previous entry.
|
||||
uint16_t HostPCOffset;
|
||||
|
||||
// How much to offset the RIP from the previous entry.
|
||||
uint16_t GuestRIPOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -5,5 +5,9 @@ namespace FEXCore::CPUID {
|
||||
struct FunctionResults {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
};
|
||||
|
||||
struct XCRResults {
|
||||
uint32_t eax, edx;
|
||||
};
|
||||
}
|
||||
|
||||
+21
-14
@@ -225,17 +225,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) = 0;
|
||||
|
||||
/**
|
||||
* @brief Sets up memory regions on the guest for mirroring within the guest's VM space
|
||||
*
|
||||
* @param VirtualAddress The address we want to set to mirror a physical memory region
|
||||
* @param PhysicalAddress The physical memory region we are mapping
|
||||
* @param Size Size of the region to mirror
|
||||
*
|
||||
* @return true when successfully mapped. false if there was an error adding
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Retrieves a feature struct indicating certain supported aspects from
|
||||
* the hose.
|
||||
@@ -254,11 +243,13 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void StopThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
|
||||
@@ -270,8 +261,8 @@ namespace FEXCore::Context {
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
@@ -288,6 +279,22 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void IncrementIdleRefCount() = 0;
|
||||
|
||||
/**
|
||||
* @brief Informs the context if hardware TSO is supported.
|
||||
* Once hardware TSO is enabled, then TSO emulation through atomics is disabled and relies on the hardware.
|
||||
*
|
||||
* @param HardwareTSOSupported If the hardware supports the TSO memory model or not.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetHardwareTSOSupport(bool HardwareTSOSupported) = 0;
|
||||
|
||||
/**
|
||||
* @brief Enable exiting the JIT when HLT is hit.
|
||||
*
|
||||
* This is to workaround a bug in Wine's longjump function which breaks our unittests.
|
||||
*
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void EnableExitOnHLT() = 0;
|
||||
private:
|
||||
};
|
||||
|
||||
|
||||
@@ -6,10 +6,69 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
// Wrapper around std::atomic using std::memory_order_relaxed.
|
||||
// This allows compilers to emit more performant code at the expense of visibly tearing.
|
||||
// In particular, increments/decrements may visibly tear if a signal is received half-way through.
|
||||
//
|
||||
// Prefer std::atomic with default memory ordering unless you really know what you're doing.
|
||||
// Primarily this ensure program ordering when signals are concerned.
|
||||
template<typename T>
|
||||
class NonAtomicRefCounter {
|
||||
public:
|
||||
void Increment(T Value) {
|
||||
// Specifically avoiding fetch_add here because that will turn in to ldxr+stxr or lock xadd.
|
||||
// FEX very specifically wants to use simple loadstore instructions for this
|
||||
//
|
||||
// ARM64 ex:
|
||||
// ldr x0, [x1];
|
||||
// add x0, x0, #1;
|
||||
// str x0, [x1];
|
||||
//
|
||||
// x86-64 ex:
|
||||
// inc qword [rax];
|
||||
auto Current = AtomicVariable.load(std::memory_order_relaxed);
|
||||
AtomicVariable.store(Current + Value, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
// Returns original value.
|
||||
// x86-64 needs to know the result on decrement.
|
||||
T Decrement(T Value) {
|
||||
// Specifically avoiding fetch_sub here because that will turn into ldxr+stxr or lock xadd.
|
||||
// FEX very specifically wants to use simple loadstore instructions for this
|
||||
//
|
||||
// ARM64 ex:
|
||||
// ldr x0, [x1];
|
||||
// sub x0, x0, #1;
|
||||
// str x0, [x1];
|
||||
//
|
||||
// x86-64 ex:
|
||||
// dec qword [rax];
|
||||
auto Current = AtomicVariable.load(std::memory_order_relaxed);
|
||||
AtomicVariable.store(Current - Value, std::memory_order_relaxed);
|
||||
return Current;
|
||||
}
|
||||
|
||||
T Load() const {
|
||||
return AtomicVariable.load(std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
void Store(T Value) {
|
||||
AtomicVariable.store(Value, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
private:
|
||||
std::atomic<T> AtomicVariable;
|
||||
};
|
||||
static_assert(std::is_standard_layout_v<NonAtomicRefCounter<uint64_t>>, "Needs to be standard layout");
|
||||
static_assert(std::is_trivially_copyable_v<NonAtomicRefCounter<uint64_t>>, "needs to be trivially copyable");
|
||||
static_assert(sizeof(NonAtomicRefCounter<uint64_t>) == sizeof(uint64_t), "Needs to be correct size");
|
||||
|
||||
struct FEX_PACKED CPUState {
|
||||
// Allows more efficient handling of the register
|
||||
// file in the event AVX is not supported.
|
||||
@@ -49,6 +108,13 @@ namespace FEXCore::Core {
|
||||
uint16_t FCW;
|
||||
uint16_t FTW;
|
||||
|
||||
uint32_t _pad2[1];
|
||||
// Reference counter for FEX's per-thread deferred signals.
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
// Since this memory region is thread local, we use NonAtomicRefCounter for fast atomic access.
|
||||
NonAtomicRefCounter<uint64_t> *DeferredSignalFaultAddress;
|
||||
|
||||
static constexpr size_t FLAG_SIZE = sizeof(flags[0]);
|
||||
static constexpr size_t GDT_SIZE = sizeof(gdt[0]);
|
||||
static constexpr size_t GPR_REG_SIZE = sizeof(gregs[0]);
|
||||
@@ -63,9 +129,29 @@ namespace FEXCore::Core {
|
||||
static constexpr size_t NUM_GPRS = sizeof(gregs) / GPR_REG_SIZE;
|
||||
static constexpr size_t NUM_XMMS = sizeof(xmm) / XMM_AVX_REG_SIZE;
|
||||
static constexpr size_t NUM_MMS = sizeof(mm) / MM_REG_SIZE;
|
||||
CPUState() {
|
||||
// Initialize default CPU state
|
||||
rip = ~0ULL;
|
||||
memset(gregs, 0, sizeof(gregs));
|
||||
|
||||
for (auto& xmm : xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(&flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
flags[1] = 1; ///< Reserved - Always 1.
|
||||
flags[9] = 1; ///< Interrupt flag - Always 1.
|
||||
FCW = 0x37F;
|
||||
FTW = 0xFFFF;
|
||||
}
|
||||
};
|
||||
static_assert(std::is_trivially_copyable_v<CPUState>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<CPUState>, "This needs to be standard layout");
|
||||
static_assert(offsetof(CPUState, xmm) % 32 == 0, "xmm needs to be 256-bit aligned!");
|
||||
static_assert(offsetof(CPUState, mm) % 16 == 0, "mm needs to be 128-bit aligned!");
|
||||
static_assert(offsetof(CPUState, DeferredSignalRefCount) % 8 == 0, "Needs to be 8-byte aligned");
|
||||
|
||||
struct InternalThreadState;
|
||||
|
||||
@@ -144,6 +230,7 @@ namespace FEXCore::Core {
|
||||
uint64_t ThreadRemoveCodeEntryFromJIT{};
|
||||
uint64_t CPUIDObj{};
|
||||
uint64_t CPUIDFunction{};
|
||||
uint64_t XCRFunction{};
|
||||
uint64_t SyscallHandlerObj{};
|
||||
uint64_t SyscallHandlerFunc{};
|
||||
uint64_t ExitFunctionLink{};
|
||||
|
||||
@@ -21,6 +21,18 @@ namespace Core {
|
||||
Return,
|
||||
ReturnRT,
|
||||
};
|
||||
|
||||
enum SignalNumber {
|
||||
#ifndef _WIN32
|
||||
FAULT_SIGSEGV = SIGSEGV,
|
||||
FAULT_SIGTRAP = SIGTRAP,
|
||||
FAULT_SIGILL = SIGILL,
|
||||
#else
|
||||
FAULT_SIGSEGV = 11,
|
||||
FAULT_SIGTRAP = 5,
|
||||
FAULT_SIGILL = 4,
|
||||
#endif
|
||||
};
|
||||
}
|
||||
using HostSignalDelegatorFunction = std::function<bool(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext)>;
|
||||
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
#pragma once
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct RuntimeStats;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
|
||||
namespace Debug {
|
||||
|
||||
void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP);
|
||||
|
||||
uint64_t GetThreadCount(FEXCore::Context::Context *CTX);
|
||||
FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(FEXCore::Context::Context *CTX, uint64_t Thread);
|
||||
|
||||
bool GetDebugDataForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, uint8_t **Code);
|
||||
// XXX:
|
||||
// bool FindIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir);
|
||||
// void SetIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -38,12 +38,6 @@ namespace FEXCore::IR{
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
|
||||
struct RuntimeStats {
|
||||
std::atomic_uint64_t InstructionsExecuted;
|
||||
std::atomic_uint64_t BlocksCompiled;
|
||||
};
|
||||
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
@@ -103,8 +97,6 @@ namespace FEXCore::Core {
|
||||
fextl::unique_ptr<FEXCore::IR::PassManager> PassManager;
|
||||
FEXCore::HLE::ThreadManagement ThreadManager;
|
||||
|
||||
RuntimeStats Stats{};
|
||||
|
||||
int StatusCode{};
|
||||
FEXCore::Context::ExitReason ExitReason {FEXCore::Context::ExitReason::EXIT_WAITING};
|
||||
std::shared_ptr<FEXCore::CompileService> CompileService;
|
||||
@@ -112,8 +104,19 @@ namespace FEXCore::Core {
|
||||
std::shared_mutex ObjectCacheRefCounter{};
|
||||
bool DestroyedByParent{false}; // Should the parent destroy this thread, or it destory itself
|
||||
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
|
||||
struct DeferredSignalState {
|
||||
#ifndef _WIN32
|
||||
siginfo_t Info;
|
||||
#endif
|
||||
int Signal;
|
||||
};
|
||||
|
||||
// Queue of thread local signal frames that have been deferred.
|
||||
// Async signals aren't guaranteed to be delivered in any particular order, but FEX treats them as FILO.
|
||||
fextl::vector<DeferredSignalState> DeferredSignalFrames;
|
||||
|
||||
// BaseFrameState should always be at the end.
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
|
||||
};
|
||||
// static_assert(std::is_standard_layout<InternalThreadState>::value, "This needs to be standard layout");
|
||||
}
|
||||
|
||||
+4
-8
@@ -3,7 +3,6 @@
|
||||
#include <shared_mutex>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXHeaderUtils/ScopedSignalMask.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
@@ -52,9 +51,8 @@ namespace FEXCore::HLE {
|
||||
class SourcecodeResolver;
|
||||
|
||||
struct AOTIRCacheEntryLookupResult {
|
||||
AOTIRCacheEntryLookupResult(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart, FHU::ScopedSignalMaskWithSharedLock &&lk)
|
||||
: Entry(Entry), VAFileStart(VAFileStart), lk(std::move(lk))
|
||||
{
|
||||
AOTIRCacheEntryLookupResult(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart)
|
||||
: Entry(Entry), VAFileStart(VAFileStart) {
|
||||
|
||||
}
|
||||
|
||||
@@ -64,8 +62,6 @@ namespace FEXCore::HLE {
|
||||
uintptr_t VAFileStart;
|
||||
|
||||
friend class SyscallHandler;
|
||||
protected:
|
||||
FHU::ScopedSignalMaskWithSharedLock lk;
|
||||
};
|
||||
|
||||
class SyscallHandler {
|
||||
@@ -78,8 +74,8 @@ namespace FEXCore::HLE {
|
||||
|
||||
SyscallOSABI GetOSABI() const { return OSABI; }
|
||||
virtual FEXCore::CodeLoader *GetCodeLoader() const { return nullptr; }
|
||||
virtual void MarkGuestExecutableRange(uint64_t Start, uint64_t Length) { }
|
||||
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(uint64_t GuestAddr) = 0;
|
||||
virtual void MarkGuestExecutableRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) { }
|
||||
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestAddr) = 0;
|
||||
|
||||
virtual SourcecodeResolver *GetSourcecodeResolver() { return nullptr; }
|
||||
protected:
|
||||
|
||||
+41
-7
@@ -4,7 +4,7 @@
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXHeaderUtils/EnumOperators.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
|
||||
#include <array>
|
||||
#include <cassert>
|
||||
@@ -332,8 +332,10 @@ static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
|
||||
struct RegisterClassType final {
|
||||
uint32_t Val;
|
||||
[[nodiscard]] constexpr operator uint32_t() const {
|
||||
using value_type = uint32_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]] friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
@@ -388,8 +390,10 @@ struct TypeDefinition final {
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
using value_type = uint8_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]] friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
|
||||
@@ -491,10 +495,20 @@ protected:
|
||||
|
||||
enum class SyscallFlags : uint8_t {
|
||||
DEFAULT = 0,
|
||||
// Syscalldoesn't care about CPUState being serialized up to the syscall instruction.
|
||||
// Means DeadCodeElimination can optimize through a syscall operation.
|
||||
OPTIMIZETHROUGH = 1 << 0,
|
||||
// Syscall only reads the passed in arguments. Doesn't read CPUState.
|
||||
NOSYNCSTATEONENTRY = 1 << 1,
|
||||
// Syscall doesn't return. Code generation after syscall return can be removed.
|
||||
NORETURN = 1 << 2,
|
||||
NOSIDEEFFECTS = 1 << 3
|
||||
// Syscall doesn't have any side-effects, so if the result isn't used then it can be removed.
|
||||
NOSIDEEFFECTS = 1 << 3,
|
||||
// Syscall doesn't return a result.
|
||||
// Means the resulting register shouldn't be written (Usually RAX).
|
||||
// Usually used with !NOSYNCSTATEONENTRY, so the syscall can modify CPU state entirely.
|
||||
// Then on return FEXCore picks up the new state.
|
||||
NORETURNEDRESULT = 1 << 4,
|
||||
};
|
||||
|
||||
FEX_DEF_NUM_OPS(SyscallFlags)
|
||||
@@ -592,7 +606,27 @@ struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID:
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) {
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
|
||||
return Base::format(ID.Value, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
|
||||
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
|
||||
return Base::format(Class.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
|
||||
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
|
||||
return Base::format(Fence.Val, ctx);
|
||||
}
|
||||
};
|
||||
+8
-1
@@ -112,7 +112,14 @@ namespace FEXCore::Allocator {
|
||||
inline void *malloc(size_t size) { return ::malloc(size); }
|
||||
inline void *calloc(size_t n, size_t size) { return ::calloc(n, size); }
|
||||
inline void *memalign(size_t align, size_t s) { return ::memalign(align, s); }
|
||||
inline void *valloc(size_t size) { return ::valloc(size); }
|
||||
inline void *valloc(size_t size)
|
||||
{
|
||||
#ifdef __ANDROID__
|
||||
return ::aligned_alloc(4096, size);
|
||||
#else
|
||||
return ::valloc(size);
|
||||
#endif
|
||||
}
|
||||
inline int posix_memalign(void** r, size_t a, size_t s) { return ::posix_memalign(r, a, s); }
|
||||
inline void *realloc(void* ptr, size_t size) { return ::realloc(ptr, size); }
|
||||
inline void free(void* ptr) { return ::free(ptr); }
|
||||
|
||||
@@ -27,6 +27,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
constexpr uint32_t LDAR_INST = 0x08'DF'FC'00;
|
||||
constexpr uint32_t LDAPR_INST = 0x38'BF'C0'00;
|
||||
constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
using ForkableUniqueMutex = std::mutex;
|
||||
using ForkableSharedMutex = std::shared_mutex;
|
||||
#endif
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedDeferredSignalWithMutexBase(const ScopedDeferredSignalWithMutexBase&) = delete;
|
||||
ScopedDeferredSignalWithMutexBase& operator=(ScopedDeferredSignalWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedDeferredSignalWithMutexBase(ScopedDeferredSignalWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex}
|
||||
, Thread {rhs.Thread} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedDeferredSignalWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the recount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedDeferredSignalWithMutex = ScopedDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedDeferredSignalWithSharedLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithUniqueLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedDeferredSignalWithForkableMutex = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedDeferredSignalWithForkableSharedLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithForkableUniqueLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedPotentialDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedPotentialDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
if (Thread) {
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
else {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
}
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedPotentialDeferredSignalWithMutexBase(const ScopedPotentialDeferredSignalWithMutexBase&) = delete;
|
||||
ScopedPotentialDeferredSignalWithMutexBase& operator=(ScopedPotentialDeferredSignalWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedPotentialDeferredSignalWithMutexBase(ScopedPotentialDeferredSignalWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex}
|
||||
, Thread {rhs.Thread} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedPotentialDeferredSignalWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
uint64_t OriginalMask{};
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedPotentialDeferredSignalWithMutex = ScopedPotentialDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithSharedLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedPotentialDeferredSignalWithForkableMutex = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithForkableSharedLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithForkableUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
File renamed without changes.
+256
@@ -0,0 +1,256 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/allocator.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#include <windows.h>
|
||||
#undef ERROR
|
||||
#endif
|
||||
|
||||
namespace FEXCore::File {
|
||||
enum class FileModes : uint32_t {
|
||||
READ = (1U << 0),
|
||||
WRITE = (1U << 1),
|
||||
CREATE = (1U << 2),
|
||||
TRUNCATE = (1U << 3),
|
||||
};
|
||||
|
||||
enum class SeekOp {
|
||||
BEGIN,
|
||||
CURRENT,
|
||||
END,
|
||||
};
|
||||
|
||||
FEX_DEF_NUM_OPS(FileModes)
|
||||
|
||||
class File final {
|
||||
public:
|
||||
#ifndef _WIN32
|
||||
using FileHandleType = int;
|
||||
#else
|
||||
using FileHandleType = HANDLE;
|
||||
#endif
|
||||
|
||||
File() = default;
|
||||
|
||||
File(const char *Filepath, FileModes Modes) {
|
||||
#ifndef _WIN32
|
||||
auto Disp = TranslateModes(Modes);
|
||||
Handle = open(Filepath, Disp, DEFAULT_USER_PERMS);
|
||||
IsValidHandle = Handle != -1;
|
||||
#else
|
||||
auto Disp = TranslateModes(Modes);
|
||||
if (Disp.CreationFlag == OPEN_ALWAYS && Disp.TruncateOnExist) {
|
||||
// If Open + Truncate then try to open with truncate behaviour first.
|
||||
Handle = CreateFileA(Filepath, Disp.Access, DEFAULT_SHARE_MODE, nullptr, TRUNCATE_EXISTING, FILE_ATTRIBUTE_NORMAL, nullptr);
|
||||
if (Handle == INVALID_HANDLE_VALUE && GetLastError() == ERROR_FILE_NOT_FOUND) {
|
||||
// File didn't exist, just open.
|
||||
Handle = CreateFileA(Filepath, Disp.Access, DEFAULT_SHARE_MODE, nullptr, CREATE_NEW, FILE_ATTRIBUTE_NORMAL, nullptr);
|
||||
}
|
||||
}
|
||||
else {
|
||||
Handle = CreateFileA(Filepath, Disp.Access, DEFAULT_SHARE_MODE, nullptr, Disp.CreationFlag, FILE_ATTRIBUTE_NORMAL, nullptr);
|
||||
}
|
||||
IsValidHandle = Handle != INVALID_HANDLE_VALUE;
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Write Bytes to File
|
||||
*
|
||||
* @param Buffer The buffer to write.
|
||||
* @param Bytes The number of bytes to write.
|
||||
*
|
||||
* @return The number of bytes actually written or -1 on error.
|
||||
*/
|
||||
ssize_t Write(void const* Buffer, size_t Bytes) {
|
||||
#ifndef _WIN32
|
||||
return write(Handle, Buffer, Bytes);
|
||||
#else
|
||||
DWORD BytesWritten{};
|
||||
auto Result = WriteFile(Handle, Buffer, Bytes, &BytesWritten, nullptr);
|
||||
if (Result) {
|
||||
return BytesWritten;
|
||||
}
|
||||
// Some error, match Linux side.
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Read at most Bytes in to the buffer.
|
||||
*
|
||||
* @param Buffer The buffer where the data is read in to.
|
||||
* @param Bytes The size of the buffer.
|
||||
*
|
||||
* @return The number of bytes read or -1 on error.
|
||||
*/
|
||||
ssize_t Read(void *Buffer, size_t Bytes) {
|
||||
#ifndef _WIN32
|
||||
return read(Handle, Buffer, Bytes);
|
||||
#else
|
||||
DWORD BytesRead{};
|
||||
auto Result = ReadFile(Handle, Buffer, Bytes, &BytesRead, nullptr);
|
||||
if (Result) {
|
||||
return BytesRead;
|
||||
}
|
||||
// Some error, match Linux side.
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
|
||||
~File() {
|
||||
if (!IsValidHandle) return;
|
||||
if (!ShouldClose) return;
|
||||
#ifndef _WIN32
|
||||
close(Handle);
|
||||
#else
|
||||
CloseHandle(Handle);
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Gets a File object that points to stdout
|
||||
*/
|
||||
static File GetStdOUT() {
|
||||
#ifndef _WIN32
|
||||
return File(STDOUT_FILENO, false);
|
||||
#else
|
||||
return File(GetStdHandle(STD_OUTPUT_HANDLE), false);
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Gets a File object that points to stderr
|
||||
*/
|
||||
static File GetStdERR() {
|
||||
#ifndef _WIN32
|
||||
return File(STDERR_FILENO, false);
|
||||
#else
|
||||
return File(GetStdHandle(STD_ERROR_HANDLE), false);
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Returns if the file handle is valid.
|
||||
*/
|
||||
bool IsValid() const { return IsValidHandle; }
|
||||
|
||||
/**
|
||||
* @brief Flush the file contents to the output file backing.
|
||||
*
|
||||
* @return True if the flush occured.
|
||||
*/
|
||||
bool Flush() {
|
||||
#ifndef _WIN32
|
||||
return fsync(Handle) == 0;
|
||||
#else
|
||||
return FlushFileBuffers(Handle);
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Seek the file pointer location.
|
||||
*
|
||||
* @param Distance The distance to travel.
|
||||
* @param Op The operation from where to start the travel.
|
||||
*
|
||||
* @return The current file pointer location or -1.
|
||||
*/
|
||||
ssize_t Seek(ssize_t Distance, SeekOp Op) {
|
||||
#ifndef _WIN32
|
||||
return lseek(Handle, Distance, TranslateSeek(Op));
|
||||
#else
|
||||
LARGE_INTEGER NewDistance {
|
||||
.QuadPart = Distance
|
||||
};
|
||||
LARGE_INTEGER NewPointer;
|
||||
auto Result = SetFilePointerEx(Handle, NewDistance, &NewPointer, TranslateSeek(Op));
|
||||
if (Result) {
|
||||
return NewPointer.QuadPart;
|
||||
}
|
||||
// Some error, match Linux side.
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
|
||||
protected:
|
||||
|
||||
File(FileHandleType Handle, bool ShouldClose)
|
||||
: ShouldClose {ShouldClose}
|
||||
, IsValidHandle {true}
|
||||
, Handle {Handle}
|
||||
{}
|
||||
private:
|
||||
bool ShouldClose{};
|
||||
bool IsValidHandle{};
|
||||
|
||||
FileHandleType Handle;
|
||||
#ifndef _WIN32
|
||||
static constexpr int DEFAULT_USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
|
||||
static uint32_t TranslateModes(FileModes Modes) {
|
||||
uint32_t Mode{};
|
||||
if ((Modes & FileModes::READ) == FileModes::READ)
|
||||
Mode |= O_RDONLY;
|
||||
if ((Modes & FileModes::WRITE) == FileModes::WRITE)
|
||||
Mode |= O_WRONLY;
|
||||
if ((Modes & FileModes::CREATE) == FileModes::CREATE)
|
||||
Mode |= O_CREAT;
|
||||
if ((Modes & FileModes::TRUNCATE) == FileModes::TRUNCATE)
|
||||
Mode |= O_TRUNC;
|
||||
|
||||
// Always enable CLOEXEC so that the FD is closed on execve.
|
||||
// FEXCore never wants to leak FDs across execve using this interface.
|
||||
Mode |= O_CLOEXEC;
|
||||
return Mode;
|
||||
}
|
||||
|
||||
static uint32_t TranslateSeek(SeekOp Op) {
|
||||
switch (Op) {
|
||||
case SeekOp::BEGIN: return SEEK_SET;
|
||||
case SeekOp::CURRENT: return SEEK_CUR;
|
||||
case SeekOp::END: return SEEK_END;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
#else
|
||||
static constexpr int DEFAULT_SHARE_MODE = FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE;
|
||||
struct Disposition {
|
||||
uint32_t CreationFlag;
|
||||
uint32_t Access;
|
||||
bool TruncateOnExist;
|
||||
};
|
||||
static Disposition TranslateModes(FileModes Modes) {
|
||||
Disposition Disp{};
|
||||
if ((Modes & FileModes::READ) == FileModes::READ)
|
||||
Disp.Access |= GENERIC_READ;
|
||||
if ((Modes & FileModes::WRITE) == FileModes::WRITE)
|
||||
Disp.Access |= GENERIC_WRITE;
|
||||
if ((Modes & FileModes::CREATE) == FileModes::CREATE)
|
||||
Disp.CreationFlag = CREATE_ALWAYS;
|
||||
else
|
||||
Disp.CreationFlag = OPEN_ALWAYS;
|
||||
|
||||
if ((Modes & FileModes::TRUNCATE) == FileModes::TRUNCATE)
|
||||
Disp.TruncateOnExist = true;
|
||||
|
||||
return Disp;
|
||||
}
|
||||
|
||||
static uint32_t TranslateSeek(SeekOp Op) {
|
||||
switch (Op) {
|
||||
case SeekOp::BEGIN: return FILE_BEGIN;
|
||||
case SeekOp::CURRENT: return FILE_CURRENT;
|
||||
case SeekOp::END: return FILE_END;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
};
|
||||
}
|
||||
@@ -37,6 +37,7 @@ namespace FEXCore::Telemetry {
|
||||
TYPE_CAS_32BIT_TEAR,
|
||||
TYPE_CAS_64BIT_TEAR,
|
||||
TYPE_CAS_128BIT_TEAR,
|
||||
TYPE_CRASH_MASK,
|
||||
TYPE_LAST,
|
||||
};
|
||||
|
||||
|
||||
+24
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/allocator.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
#include <unistd.h>
|
||||
@@ -38,6 +39,7 @@ namespace fextl::fmt {
|
||||
return fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
template <typename... T>
|
||||
FMT_INLINE auto print(::fmt::format_string<T...> fmt, T&&... args)
|
||||
-> void {
|
||||
@@ -51,6 +53,28 @@ namespace fextl::fmt {
|
||||
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
write(FD, String.c_str(), String.size());
|
||||
}
|
||||
#else
|
||||
template <typename... T>
|
||||
FMT_INLINE auto print(::fmt::format_string<T...> fmt, T&&... args)
|
||||
-> void {
|
||||
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
auto f = fextl::file::File::GetStdOUT();
|
||||
f.Write(String.c_str(), String.size());
|
||||
}
|
||||
|
||||
template <typename... T>
|
||||
FMT_INLINE auto print(HANDLE File, ::fmt::format_string<T...> fmt, T&&... args)
|
||||
-> void {
|
||||
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
WriteFile(File, String.c_str(), String.size(), nullptr, nullptr);
|
||||
}
|
||||
#endif
|
||||
template <typename... T>
|
||||
FMT_INLINE auto print(FEXCore::File::File& f, ::fmt::format_string<T...> fmt, T&&... args)
|
||||
-> void {
|
||||
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
f.Write(String.c_str(), String.size());
|
||||
}
|
||||
|
||||
template <typename... T>
|
||||
FMT_INLINE auto print(std::FILE* f, ::fmt::format_string<T...> fmt, T&&... args)
|
||||
|
||||
+3
-1
@@ -1 +1,3 @@
|
||||
add_subdirectory(Emitter/)
|
||||
if (NOT MINGW_BUILD)
|
||||
add_subdirectory(Emitter/)
|
||||
endif()
|
||||
+44
-42
@@ -301,6 +301,33 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Add/subtract immediate") {
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r28, 4095, true), "cmp x28, #0xfff000 (16773120)");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r28, 16773120), "cmp x28, #0xfff000 (16773120)");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Min/max immediate") {
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, 1), "smax w29, w28, #1");
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, 127), "smax w29, w28, #127");
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, -128), "smax w29, w28, #-128");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, 1), "smax x29, x28, #1");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, 127), "smax x29, x28, #127");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, -128), "smax x29, x28, #-128");
|
||||
|
||||
TEST_SINGLE(umax(Size::i32Bit, Reg::r29, Reg::r28, 0), "umax w29, w28, #0");
|
||||
TEST_SINGLE(umax(Size::i32Bit, Reg::r29, Reg::r28, 255), "umax w29, w28, #255");
|
||||
TEST_SINGLE(umax(Size::i64Bit, Reg::r29, Reg::r28, 0), "umax x29, x28, #0");
|
||||
TEST_SINGLE(umax(Size::i64Bit, Reg::r29, Reg::r28, 255), "umax x29, x28, #255");
|
||||
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, 1), "smin w29, w28, #1");
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, 127), "smin w29, w28, #127");
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, -128), "smin w29, w28, #-128");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, 1), "smin x29, x28, #1");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, 127), "smin x29, x28, #127");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, -128), "smin x29, x28, #-128");
|
||||
|
||||
TEST_SINGLE(umin(Size::i32Bit, Reg::r29, Reg::r28, 0), "umin w29, w28, #0");
|
||||
TEST_SINGLE(umin(Size::i32Bit, Reg::r29, Reg::r28, 255), "umin w29, w28, #255");
|
||||
TEST_SINGLE(umin(Size::i64Bit, Reg::r29, Reg::r28, 0), "umin x29, x28, #0");
|
||||
TEST_SINGLE(umin(Size::i64Bit, Reg::r29, Reg::r28, 255), "umin x29, x28, #255");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Logical immediate") {
|
||||
TEST_SINGLE(and_(Size::i32Bit, Reg::r29, Reg::r28, 1), "and w29, w28, #0x1");
|
||||
TEST_SINGLE(and_(Size::i32Bit, Reg::r29, Reg::r28, -2), "and w29, w28, #0xfffffffe");
|
||||
@@ -428,6 +455,10 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Data processing - 2 source") {
|
||||
TEST_SINGLE(crc32cb(WReg::w29, WReg::w28, WReg::w27), "crc32cb w29, w28, w27");
|
||||
TEST_SINGLE(crc32ch(WReg::w29, WReg::w28, WReg::w27), "crc32ch w29, w28, w27");
|
||||
TEST_SINGLE(crc32cw(WReg::w29, WReg::w28, WReg::w27), "crc32cw w29, w28, w27");
|
||||
TEST_SINGLE(smax(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "smax w29, w28, w27");
|
||||
TEST_SINGLE(umax(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "umax w29, w28, w27");
|
||||
TEST_SINGLE(smin(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "smin w29, w28, w27");
|
||||
TEST_SINGLE(umin(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "umin w29, w28, w27");
|
||||
|
||||
TEST_SINGLE(udiv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "udiv x29, x28, x27");
|
||||
TEST_SINGLE(sdiv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "sdiv x29, x28, x27");
|
||||
@@ -435,6 +466,10 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Data processing - 2 source") {
|
||||
TEST_SINGLE(lsrv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "lsr x29, x28, x27");
|
||||
TEST_SINGLE(asrv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "asr x29, x28, x27");
|
||||
TEST_SINGLE(rorv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "ror x29, x28, x27");
|
||||
TEST_SINGLE(smax(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "smax x29, x28, x27");
|
||||
TEST_SINGLE(umax(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "umax x29, x28, x27");
|
||||
TEST_SINGLE(smin(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "smin x29, x28, x27");
|
||||
TEST_SINGLE(umin(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "umin x29, x28, x27");
|
||||
|
||||
if (false) {
|
||||
// vixl doesn't support this instruction.
|
||||
@@ -471,6 +506,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Data processing - 1 source") {
|
||||
TEST_SINGLE(rev(XReg::x29, XReg::x28), "rev x29, x28");
|
||||
TEST_SINGLE(rev(Size::i32Bit, Reg::r29, Reg::r28), "rev w29, w28");
|
||||
TEST_SINGLE(rev(Size::i64Bit, Reg::r29, Reg::r28), "rev x29, x28");
|
||||
|
||||
TEST_SINGLE(ctz(Size::i32Bit, Reg::r29, Reg::r28), "ctz w29, w28");
|
||||
TEST_SINGLE(ctz(Size::i64Bit, Reg::r29, Reg::r28), "ctz x29, x28");
|
||||
|
||||
TEST_SINGLE(cnt(Size::i32Bit, Reg::r29, Reg::r28), "cnt w29, w28");
|
||||
TEST_SINGLE(cnt(Size::i64Bit, Reg::r29, Reg::r28), "cnt x29, x28");
|
||||
|
||||
TEST_SINGLE(abs(Size::i32Bit, Reg::r29, Reg::r28), "abs w29, w28");
|
||||
TEST_SINGLE(abs(Size::i64Bit, Reg::r29, Reg::r28), "abs x29, x28");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PAUTH") {
|
||||
// TODO: Implement in the emitter.
|
||||
@@ -782,24 +826,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "add x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "add w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "add x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "add w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "add x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "add w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "add x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "add w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "add x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "add w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(add(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -814,24 +852,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "adds x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "adds w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "adds x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "adds w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "adds x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "adds w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "adds x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "adds w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "adds x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "adds w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -849,24 +881,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "sub x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "sub w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "sub x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "sub w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "sub x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "sub w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "sub x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "sub w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "sub x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "sub w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -881,24 +907,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 63), "subs x30, x29, x28, lsl #63");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 31), "subs w30, w29, w28, lsl #31");
|
||||
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "subs x30, x29, x28, lsr #1");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 1), "subs w30, w29, w28, lsr #1");
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 63), "subs x30, x29, x28, lsr #63");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 31), "subs w30, w29, w28, lsr #31");
|
||||
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "subs x30, x29, x28, asr #1");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 1), "subs w30, w29, w28, asr #1");
|
||||
TEST_SINGLE(subs(Size::i64Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 63), "subs x30, x29, x28, asr #63");
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 31), "subs w30, w29, w28, asr #31");
|
||||
|
||||
TEST_SINGLE(subs(Size::i32Bit, Reg::r30, Reg::r29, Reg::r28, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -913,24 +933,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSL, 63), "neg x30, x29, lsl #63");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 31), "neg w30, w29, lsl #31");
|
||||
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "neg x30, x29, lsr #1");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "neg w30, w29, lsr #1");
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 63), "neg x30, x29, lsr #63");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 31), "neg w30, w29, lsr #31");
|
||||
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "neg x30, x29, asr #1");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "neg w30, w29, asr #1");
|
||||
TEST_SINGLE(neg(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 63), "neg x30, x29, asr #63");
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 31), "neg w30, w29, asr #31");
|
||||
|
||||
TEST_SINGLE(neg(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -945,24 +959,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSL, 63), "cmp x30, x29, lsl #63");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 31), "cmp w30, w29, lsl #31");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "cmp x30, x29, lsr #1");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "cmp w30, w29, lsr #1");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 63), "cmp x30, x29, lsr #63");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 31), "cmp w30, w29, lsr #31");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "cmp x30, x29, asr #1");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "cmp w30, w29, asr #1");
|
||||
TEST_SINGLE(cmp(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 63), "cmp x30, x29, asr #63");
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 31), "cmp w30, w29, asr #31");
|
||||
|
||||
TEST_SINGLE(cmp(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
@@ -977,24 +985,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSL, 63), "negs x30, x29, lsl #63");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 31), "negs w30, w29, lsl #31");
|
||||
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSL, 32), "unallocated (Unallocated)");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "negs x30, x29, lsr #1");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 1), "negs w30, w29, lsr #1");
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::LSR, 63), "negs x30, x29, lsr #63");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 31), "negs w30, w29, lsr #31");
|
||||
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::LSR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "negs x30, x29, asr #1");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 1), "negs w30, w29, asr #1");
|
||||
TEST_SINGLE(negs(Size::i64Bit, Reg::r30, Reg::r29, ShiftType::ASR, 63), "negs x30, x29, asr #63");
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 31), "negs w30, w29, asr #31");
|
||||
|
||||
TEST_SINGLE(negs(Size::i32Bit, Reg::r30, Reg::r29, ShiftType::ASR, 32), "unallocated (Unallocated)");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
|
||||
+1
-1
@@ -2110,7 +2110,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE FFR write from predicate")
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE FFR initialise") {
|
||||
TEST_SINGLE(setffr(), "setffr ");
|
||||
TEST_SINGLE(setffr(), "setffr");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE Integer Multiply-Add - Unpredicated") {
|
||||
|
||||
+8
-8
@@ -31,16 +31,16 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System Instruction") {
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGSW, Reg::r30), "sys #0, C7, C14, #4, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGDSW, Reg::r30), "sys #0, C7, C14, #6, x30");
|
||||
|
||||
TEST_SINGLE(dc(DataCacheOperation::GVA, Reg::r30), "sys #3, C7, C4, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::GZVA, Reg::r30), "sys #3, C7, C4, #4, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAC, Reg::r30), "sys #3, C7, C10, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAC, Reg::r30), "sys #3, C7, C10, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAP, Reg::r30), "sys #3, C7, C12, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAP, Reg::r30), "sys #3, C7, C12, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::GVA, Reg::r30), "dc gva, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::GZVA, Reg::r30), "dc gzva, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAC, Reg::r30), "dc cgvac, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAC, Reg::r30), "dc cgdvac, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVAP, Reg::r30), "dc cgvap, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVAP, Reg::r30), "dc cgdvap, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGVADP, Reg::r30), "sys #3, C7, C13, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CGDVADP, Reg::r30), "sys #3, C7, C13, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGVAC, Reg::r30), "sys #3, C7, C14, #3, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGDVAC, Reg::r30), "sys #3, C7, C14, #5, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGVAC, Reg::r30), "dc cigvac, x30");
|
||||
TEST_SINGLE(dc(DataCacheOperation::CIGDVAC, Reg::r30), "dc cigdvac, x30");
|
||||
|
||||
TEST_SINGLE(dc(DataCacheOperation::CVAP, Reg::r30), "dc cvap, x30");
|
||||
|
||||
|
||||
Vendored
+1
-1
Submodule External/fmt updated: c4ee726532...a0b8a92e3d.
Vendored
+1
-1
Submodule External/jemalloc updated: 19196715f1...16f8061955.
Vendored
+1
-1
Submodule External/jemalloc_glibc updated: 79e02853ed...888181c5f7.
Vendored
+1
-1
Submodule External/vixl updated: c9e5307fe9...96f22fe65d.
@@ -265,7 +265,7 @@ namespace FHU::Filesystem {
|
||||
size_t DataSize = (sizeof(std::string_view) + sizeof(void*) * 2) * (SeparatorCount + 2);
|
||||
void *Data = alloca(DataSize);
|
||||
fextl::pmr::fixed_size_monotonic_buffer_resource mbr(Data, DataSize);
|
||||
std::pmr::polymorphic_allocator pa {&mbr};
|
||||
std::pmr::polymorphic_allocator<std::byte> pa {&mbr};
|
||||
std::pmr::list<std::string_view> Parts{pa};
|
||||
|
||||
size_t CurrentOffset{};
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
@@ -106,4 +106,17 @@ namespace FHU {
|
||||
using ScopedSignalMaskWithMutex = ScopedSignalMaskWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedSignalMaskWithSharedLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithUniqueLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
using ScopedSignalMaskWithForkableMutex = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedSignalMaskWithForkableSharedLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithForkableUniqueLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -305,7 +305,7 @@ def GetRootFSPath():
|
||||
return _RootFSPath
|
||||
|
||||
def CheckRootFSInstallStatus():
|
||||
# Matches what is available on https://rootfs.fex-emu.com/file/fex-rootfs/RootFS_links.json
|
||||
# Matches what is available on https://rootfs.fex-emu.gg/RootFS_links.json
|
||||
UbuntuVersionToRootFS = {
|
||||
"20.04": "Ubuntu_20_04.sqsh",
|
||||
"20.04": "Ubuntu_20_04.ero",
|
||||
@@ -313,6 +313,8 @@ def CheckRootFSInstallStatus():
|
||||
"22.04": "Ubuntu_22_04.ero",
|
||||
"22.10": "Ubuntu_22_10.sqsh",
|
||||
"22.10": "Ubuntu_22_10.ero",
|
||||
"23.04": "Ubuntu_23_04.sqsh",
|
||||
"23.04": "Ubuntu_23_04.ero",
|
||||
}
|
||||
|
||||
return os.path.exists(GetRootFSPath() + UbuntuVersionToRootFS[GetDistro()[1]])
|
||||
|
||||
@@ -9,6 +9,7 @@ from threading import Thread
|
||||
import subprocess
|
||||
import time
|
||||
import multiprocessing
|
||||
from shutil import which
|
||||
|
||||
if sys.version_info[0] < 3:
|
||||
raise Exception("Python 3 or a more recent version is required.")
|
||||
@@ -41,6 +42,10 @@ def Threaded_Manager(Runner, ID, File):
|
||||
ServerArgs = ["catchsegv", Runner, "-c", "vm", "-n", "1", "-I", "R" + str(ID), File]
|
||||
ClientArgs = ["catchsegv", Runner, "-c", "vm", "-n", "1", "-I", "R" + str(ID), "-C"]
|
||||
|
||||
if which("catchsegv") is None:
|
||||
ServerArgs.pop(0)
|
||||
ClientArgs.pop(0)
|
||||
|
||||
ServerThread = Thread(target = Threaded_Runner, args = (ServerArgs, ID, 0))
|
||||
ClientThread = Thread(target = Threaded_Runner, args = (ClientArgs, ID, 1))
|
||||
|
||||
|
||||
@@ -73,6 +73,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_BMI1 = (1 << 6)
|
||||
FEATURE_BMI2 = (1 << 7)
|
||||
FEATURE_CLWB = (1 << 8)
|
||||
FEATURE_LINUX = (1 << 9)
|
||||
|
||||
RegStringLookup = {
|
||||
"NONE": Regs.REG_NONE,
|
||||
@@ -145,6 +146,7 @@ HostFeaturesLookup = {
|
||||
"BMI1" : HostFeatures.FEATURE_BMI1,
|
||||
"BMI2" : HostFeatures.FEATURE_BMI2,
|
||||
"CLWB" : HostFeatures.FEATURE_CLWB,
|
||||
"LINUX" : HostFeatures.FEATURE_LINUX,
|
||||
}
|
||||
|
||||
def parse_hexstring(s):
|
||||
|
||||
@@ -3,6 +3,7 @@ import sys
|
||||
import subprocess
|
||||
import os.path
|
||||
from os import path
|
||||
from shutil import which
|
||||
|
||||
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
|
||||
|
||||
@@ -46,6 +47,9 @@ if path.exists(disabled_tests_runner_file):
|
||||
disabled_tests[line.strip()] = 1
|
||||
|
||||
RunnerArgs = ["catchsegv", runner]
|
||||
|
||||
if which("catchsegv") is None:
|
||||
RunnerArgs.pop(0)
|
||||
# Add the rest of the arguments
|
||||
for i in range(len(sys.argv) - args_start_index):
|
||||
RunnerArgs.append(sys.argv[args_start_index + i])
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
add_subdirectory(Common/)
|
||||
add_subdirectory(Tools/)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
add_subdirectory(Common/)
|
||||
add_subdirectory(Linux/)
|
||||
add_subdirectory(Tools/)
|
||||
endif()
|
||||
@@ -3,13 +3,17 @@ add_subdirectory(cpp-optparse/)
|
||||
set(NAME Common)
|
||||
set(SRCS
|
||||
ArgumentLoader.cpp
|
||||
Config.cpp
|
||||
EnvironmentLoader.cpp
|
||||
FEXServerClient.cpp
|
||||
FileFormatCheck.cpp
|
||||
StringUtil.cpp)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND SRCS
|
||||
Config.cpp
|
||||
FEXServerClient.cpp
|
||||
FileFormatCheck.cpp)
|
||||
endif()
|
||||
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse json-maker FEXHeaderUtils)
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse tiny-json json-maker FEXHeaderUtils)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/External/cpp-optparse/)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_BINARY_DIR}/generated)
|
||||
+232
-7
@@ -5,6 +5,7 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
#include <FEXHeaderUtils/SymlinkChecks.h>
|
||||
|
||||
@@ -14,8 +15,74 @@
|
||||
#include <pwd.h>
|
||||
#include <utility>
|
||||
#include <json-maker.h>
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEX::Config {
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static const fextl::map<FEXCore::Config::ConfigOption, fextl::string> ConfigToNameLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {FEXCore::Config::ConfigOption::CONFIG_##enum, #json},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
@@ -45,6 +112,164 @@ namespace FEX::Config {
|
||||
}
|
||||
}
|
||||
|
||||
// Application loaders
|
||||
class OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit OptionMapper(FEXCore::Config::LayerType Layer);
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char *ConfigName, const char *ConfigString);
|
||||
};
|
||||
|
||||
class MainLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return fextl::make_unique<MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<EnvLoader>(_envp);
|
||||
}
|
||||
|
||||
fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInterp, const std::string_view ProgramFDFromEnv) {
|
||||
// If executed with a FEX FD then the Program argument might be empty.
|
||||
// In this case we need to scan the FD node to recover the application binary that exists on disk.
|
||||
@@ -120,8 +345,8 @@ namespace FEX::Config {
|
||||
const std::string_view ProgramFDFromEnv) {
|
||||
FEX::Config::InitializeConfigs();
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(CreateMainLayer());
|
||||
|
||||
if (NoFEXArguments) {
|
||||
FEX::ArgLoader::LoadWithoutArguments(argc, argv);
|
||||
@@ -130,7 +355,7 @@ namespace FEX::Config {
|
||||
FEXCore::Config::AddLayer(fextl::make_unique<FEX::ArgLoader::ArgLoader>(argc, argv));
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::AddLayer(CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
auto Args = FEX::ArgLoader::Get();
|
||||
@@ -183,16 +408,16 @@ namespace FEX::Config {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
|
||||
auto SteamID = getenv("SteamAppId");
|
||||
if (SteamID) {
|
||||
// If a SteamID exists then let's search for Steam application configs as well.
|
||||
// We want to key off both the SteamAppId number /and/ the executable since we may not want to thunk all binaries.
|
||||
fextl::string SteamAppName = fextl::fmt::format("Steam_{}_{}", SteamID, ProgramName);
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
}
|
||||
|
||||
return ApplicationNames{std::move(Program), std::move(ProgramName)};
|
||||
|
||||
@@ -55,4 +55,40 @@ namespace FEX::Config {
|
||||
fextl::string GetConfigFileLocation(bool Global);
|
||||
|
||||
void InitializeConfigs();
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
|
||||
/**
|
||||
* @brief Loads the main application config
|
||||
*
|
||||
* @param File Optional override to load a specific config file in to the main layer
|
||||
* Shouldn't be commonly used
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
*
|
||||
* @param Filename Application filename component
|
||||
* @param Global Load the global configuration or user accessible file
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
*
|
||||
* @param _envp[] The environment array from main
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]);
|
||||
}
|
||||
@@ -153,7 +153,7 @@ namespace FEXServerClient {
|
||||
auto ServerSocketName = GetServerSocketName();
|
||||
|
||||
// Create the initial unix socket
|
||||
int SocketFD = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
int SocketFD = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (SocketFD == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
return -1;
|
||||
|
||||
+12
-15
@@ -1,13 +1,9 @@
|
||||
if (ENABLE_VISUAL_DEBUGGER)
|
||||
add_subdirectory(Debugger/)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
if (NOT TERMUX_BUILD)
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
if (BUILD_FEXCONFIG)
|
||||
add_subdirectory(FEXConfig/)
|
||||
endif()
|
||||
|
||||
if (NOT TERMUX_BUILD)
|
||||
# Disable FEXRootFSFetcher on Termux, it doesn't even work there
|
||||
add_subdirectory(FEXRootFSFetcher/)
|
||||
endif()
|
||||
@@ -19,13 +15,14 @@ if (NOT MINGW_BUILD)
|
||||
add_subdirectory(FEXGetConfig/)
|
||||
add_subdirectory(FEXServer/)
|
||||
add_subdirectory(FEXBash/)
|
||||
add_subdirectory(FEXLoader/)
|
||||
|
||||
set(NAME Opt)
|
||||
set(SRCS Opt.cpp)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} FEXCore Common pthread)
|
||||
endif()
|
||||
|
||||
set(NAME Opt)
|
||||
set(SRCS Opt.cpp)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} FEXCore Common pthread)
|
||||
add_subdirectory(FEXLoader/)
|
||||
@@ -1,32 +0,0 @@
|
||||
set(NAME Debugger)
|
||||
set(SRCS Main.cpp
|
||||
DebuggerState.cpp
|
||||
Context.cpp
|
||||
FEXImGui.cpp
|
||||
IMGui_I.cpp
|
||||
IRLexer.cpp
|
||||
GLUtils.cpp
|
||||
MainWindow.cpp
|
||||
Disassembler.cpp
|
||||
${CMAKE_SOURCE_DIR}/External/imgui/examples/imgui_impl_glfw.cpp
|
||||
${CMAKE_SOURCE_DIR}/External/imgui/examples/imgui_impl_opengl3.cpp
|
||||
)
|
||||
|
||||
find_library(EPOXY_LIBRARY epoxy)
|
||||
find_library(GLFW_LIBRARY glfw3)
|
||||
find_package(LLVM CONFIG QUIET)
|
||||
if(LLVM_FOUND AND TARGET LLVM)
|
||||
message(STATUS "LLVM found!")
|
||||
include_directories(${LLVM_INCLUDE_DIRS})
|
||||
endif()
|
||||
|
||||
add_definitions(-DIMGUI_IMPL_OPENGL_LOADER_CUSTOM=<epoxy/gl.h>)
|
||||
add_executable(${NAME} ${SRCS})
|
||||
|
||||
target_link_libraries(${NAME} PRIVATE LLVM)
|
||||
target_include_directories(${NAME} PRIVATE ${LLVM_INCLUDE_DIRS})
|
||||
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_SOURCE_DIR}/Source/)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_SOURCE_DIR}/External/imgui/examples/)
|
||||
|
||||
target_link_libraries(${NAME} PRIVATE FEXCore Common pthread LLVM epoxy glfw X11 EGL imgui tiny-json json-maker)
|
||||
@@ -1,91 +0,0 @@
|
||||
#include "Context.h"
|
||||
|
||||
#include <cassert>
|
||||
#include <GLFW/glfw3.h>
|
||||
#include <vector>
|
||||
#include <cstdio>
|
||||
|
||||
namespace GLContext {
|
||||
void glfw_error_callback(int error, const char* description)
|
||||
{
|
||||
fprintf(stderr, "Glfw Error %d: %s\n", error, description);
|
||||
}
|
||||
|
||||
class GLFWContext final : public GLContext::Context {
|
||||
public:
|
||||
void Create(const char *Title) override {
|
||||
glfwSetErrorCallback(glfw_error_callback);
|
||||
if (!glfwInit()) {
|
||||
assert(0 && "Couldn't init glfw");
|
||||
}
|
||||
glfwWindowHint(GLFW_CONTEXT_VERSION_MAJOR, 4);
|
||||
glfwWindowHint(GLFW_CONTEXT_VERSION_MINOR, 1);
|
||||
glfwWindowHint(GLFW_OPENGL_PROFILE, GLFW_OPENGL_CORE_PROFILE);
|
||||
glfwWindowHint(GLFW_OPENGL_FORWARD_COMPAT, GL_TRUE);
|
||||
glfwWindowHint(GLFW_RED_BITS, 8);
|
||||
glfwWindowHint(GLFW_GREEN_BITS, 8);
|
||||
glfwWindowHint(GLFW_BLUE_BITS, 8);
|
||||
glfwWindowHint(GLFW_ALPHA_BITS, 8);
|
||||
glfwWindowHint(GLFW_RESIZABLE, 1);
|
||||
glfwWindowHint(GLFW_DOUBLEBUFFER, 1);
|
||||
|
||||
Window = glfwCreateWindow(640, 640, Title, nullptr, nullptr);
|
||||
if (!Window) {
|
||||
assert(0 && "Couldn't create window");
|
||||
}
|
||||
glfwMakeContextCurrent(Window);
|
||||
glfwSwapInterval(1);
|
||||
}
|
||||
|
||||
void Shutdown() override {
|
||||
glfwMakeContextCurrent(nullptr);
|
||||
glfwDestroyWindow(Window);
|
||||
glfwTerminate();
|
||||
}
|
||||
|
||||
void Swap() override {
|
||||
glfwSwapBuffers(Window);
|
||||
CheckWindowDimensions();
|
||||
}
|
||||
|
||||
void RegisterResizeEvent(ResizeEvent Event) override {
|
||||
ResizeEvents.emplace_back(Event);
|
||||
}
|
||||
|
||||
void GetDim(uint32_t *Dim) override {
|
||||
Dim[0] = Width;
|
||||
Dim[1] = Height;
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
void CheckWindowDimensions() {
|
||||
int LocalWidth;
|
||||
int LocalHeight;
|
||||
glfwGetWindowSize(Window, &LocalWidth, &LocalHeight);
|
||||
|
||||
if (LocalHeight != Height || LocalWidth != Width) {
|
||||
Width = LocalWidth;
|
||||
Height = LocalHeight;
|
||||
|
||||
for (auto const &Event : ResizeEvents) {
|
||||
Event(Width, Height);
|
||||
}
|
||||
}
|
||||
}
|
||||
void* GetWindow() override {
|
||||
return Window;
|
||||
}
|
||||
|
||||
std::vector<ResizeEvent> ResizeEvents;
|
||||
|
||||
GLFWwindow *Window;
|
||||
uint32_t Width{};
|
||||
uint32_t Height{};
|
||||
};
|
||||
|
||||
std::unique_ptr<Context> CreateContext() {
|
||||
return std::make_unique<GLFWContext>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,20 +0,0 @@
|
||||
#pragma once
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
|
||||
namespace GLContext {
|
||||
class Context {
|
||||
public:
|
||||
virtual ~Context() {}
|
||||
virtual void Create(const char *Title) = 0;
|
||||
virtual void Shutdown() = 0;
|
||||
virtual void Swap() = 0;
|
||||
virtual void GetDim(uint32_t *Dim) = 0;
|
||||
|
||||
using ResizeEvent = std::function<void(uint32_t, uint32_t)>;
|
||||
virtual void RegisterResizeEvent(ResizeEvent Event) = 0;
|
||||
virtual void* GetWindow() = 0;
|
||||
};
|
||||
|
||||
std::unique_ptr<Context> CreateContext();
|
||||
}
|
||||
Loaded 100 of 179 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user