mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 19:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2f5ebf1dd1 | ||
|
|
8f157e45bb | ||
|
|
6b259e2731 | ||
|
|
200660aba5 | ||
|
|
065c12cfbb | ||
|
|
318972620f | ||
|
|
e7f54d1592 | ||
|
|
a8571282b2 | ||
|
|
fc28062052 | ||
|
|
cc6306aa32 | ||
|
|
f0caa81253 | ||
|
|
cd98871f8f | ||
|
|
66e0d46d89 | ||
|
|
1dd5642e46 | ||
|
|
9ca34ca306 | ||
|
|
69f39a0bc3 | ||
|
|
50cf74db0a | ||
|
|
6886d8ff65 | ||
|
|
c1d118c1d4 | ||
|
|
5e46d63c42 | ||
|
|
863a59a8e2 | ||
|
|
c37fcf136a | ||
|
|
02a2292115 | ||
|
|
a483bc9837 | ||
|
|
120a6b85f4 | ||
|
|
9fea774a93 | ||
|
|
bf1e619ead | ||
|
|
0f8fcfc43e | ||
|
|
698b7fda06 | ||
|
|
23caa6e20f | ||
|
|
34e39c996e | ||
|
|
16ed20cfae | ||
|
|
ef368ceafa | ||
|
|
45480ef32c | ||
|
|
4de69029e5 | ||
|
|
c065770f48 | ||
|
|
27957ea051 | ||
|
|
94e9d1ab3b | ||
|
|
a374a9af35 | ||
|
|
3e80416eb6 | ||
|
|
b35c6c6d22 | ||
|
|
69d26cfee6 | ||
|
|
e3be1540f1 | ||
|
|
8e2b0d10e5 | ||
|
|
57c5761920 | ||
|
|
0841ff5feb | ||
|
|
45115384b5 | ||
|
|
dae2563850 | ||
|
|
fb4df5a0b7 | ||
|
|
b2f0303d1e | ||
|
|
f8b2a0b4d8 | ||
|
|
e1fcb78ce3 | ||
|
|
d6309088c0 | ||
|
|
1fb3a2e28f | ||
|
|
df25d4e03e | ||
|
|
9c2f0287e0 | ||
|
|
9912d41714 | ||
|
|
8b2cd87d9e | ||
|
|
3e6d23ae7e | ||
|
|
c96c39d5b1 | ||
|
|
a4556e90cd | ||
|
|
4d2c4b4423 | ||
|
|
530de3f031 | ||
|
|
c9622f6fd4 | ||
|
|
854628d959 | ||
|
|
35dbf6c44b | ||
|
|
b9fec7436f | ||
|
|
808d19c374 | ||
|
|
441d7205ed | ||
|
|
8b3d3b68c6 | ||
|
|
aa0e038ef7 | ||
|
|
5e5e5a35d9 | ||
|
|
10e35a55ea | ||
|
|
83cebea780 | ||
|
|
91bbb92c50 | ||
|
|
c7dd6ff28a | ||
|
|
df761a99ce | ||
|
|
359416e2b6 | ||
|
|
e140c0d60c | ||
|
|
0030971f6f | ||
|
|
3a90aaf1e6 | ||
|
|
f8a199af49 | ||
|
|
3a5de8e10c | ||
|
|
3c881809f4 | ||
|
|
8b6e9e08c0 | ||
|
|
3c8da3e3b4 | ||
|
|
0a39d909b2 | ||
|
|
d35d1092a4 | ||
|
|
7ae655c56a | ||
|
|
58f35ba413 | ||
|
|
e61eb24ec2 | ||
|
|
2e93d2ce51 | ||
|
|
9d21e1efd5 | ||
|
|
e0e6b3ad6b | ||
|
|
815cdc5b3c | ||
|
|
bc20f1e684 | ||
|
|
20e5f2bec6 | ||
|
|
c3b6fa55b6 | ||
|
|
69045db3a9 | ||
|
|
7b2240c80b | ||
|
|
dcfbd90dd7 | ||
|
|
70e6ab5782 | ||
|
|
175879823f | ||
|
|
2271a90adb | ||
|
|
2bdde5845e | ||
|
|
0c40497a01 | ||
|
|
45dff0f550 | ||
|
|
e9035ef6ee | ||
|
|
02ca94e6e6 | ||
|
|
432b7d2dc8 | ||
|
|
181d315d2c | ||
|
|
d5f7e616eb | ||
|
|
9a8869e8a4 | ||
|
|
85d56ed76f | ||
|
|
06c827b5c8 | ||
|
|
41259ff361 | ||
|
|
cccee1a668 | ||
|
|
c46b35362b | ||
|
|
2f96a6d8bf | ||
|
|
c78a47a3e5 | ||
|
|
b049721683 | ||
|
|
11c06fe9fe | ||
|
|
c5f6e53d0d | ||
|
|
1184672bb1 | ||
|
|
57d3a2ba35 | ||
|
|
1d813b0183 | ||
|
|
9b24931518 | ||
|
|
7e5b8b7bdf | ||
|
|
7e36473aff | ||
|
|
adbb512306 | ||
|
|
b951f4ad4b | ||
|
|
48900662ae | ||
|
|
da31a66c07 | ||
|
|
ca710b1cbb | ||
|
|
a4d7eec145 | ||
|
|
232c2fe87f | ||
|
|
661112cfd4 | ||
|
|
d4a84eaa9a | ||
|
|
99f8af64d0 | ||
|
|
1541ea9ffc | ||
|
|
ced86e693c | ||
|
|
53920d5bd3 | ||
|
|
15c5a9dac0 | ||
|
|
d63cfdbe7d | ||
|
|
569461a01c | ||
|
|
37c7dee236 | ||
|
|
cc230091c8 | ||
|
|
bac33cf246 | ||
|
|
7a9c0506b4 | ||
|
|
fbd7c15a4b | ||
|
|
64edf24bc7 | ||
|
|
3f456d683f | ||
|
|
aac0824fd3 | ||
|
|
1b32ca0b93 | ||
|
|
c29456aac9 | ||
|
|
715f25d059 | ||
|
|
c85e31ec7a | ||
|
|
0d6bfc4fa4 | ||
|
|
106917add2 | ||
|
|
1e10a2bac3 | ||
|
|
63dd09cd6f | ||
|
|
c7e6935f42 | ||
|
|
1be5054e86 | ||
|
|
f40755aca7 | ||
|
|
d49b78cf34 | ||
|
|
10e80ae064 | ||
|
|
f3ebb214aa | ||
|
|
dc41c3bd9f | ||
|
|
56ff09f3ac | ||
|
|
1707b27d14 | ||
|
|
7c92963eca | ||
|
|
ecf82c90ee | ||
|
|
580f06fe00 | ||
|
|
5a403b7765 | ||
|
|
f7367e56af | ||
|
|
f066abc151 | ||
|
|
2bee92f332 | ||
|
|
7d9ed4e1bf | ||
|
|
24696e6b98 | ||
|
|
71f658b07d | ||
|
|
920f56353f | ||
|
|
e0fe9167ea | ||
|
|
45a2349a0d | ||
|
|
d7b0e8469e | ||
|
|
8afc3b8e23 | ||
|
|
9caa63d5b2 | ||
|
|
1d32df91ce | ||
|
|
fa973f65bf | ||
|
|
c027f02e5a | ||
|
|
9cee0126d7 | ||
|
|
c713602d56 | ||
|
|
c8293cbbda | ||
|
|
a9c51388cf | ||
|
|
ad1d65e91a | ||
|
|
3a6c7803e8 | ||
|
|
aa837eddd5 | ||
|
|
faef57838f | ||
|
|
04d4c5e017 | ||
|
|
9158877569 | ||
|
|
1eae07f1b8 | ||
|
|
14b22487f1 | ||
|
|
1ac7cd5835 | ||
|
|
5336f01725 | ||
|
|
b8b66b1829 | ||
|
|
fd3e988a20 | ||
|
|
b1d98f4e58 | ||
|
|
9e7daf61d0 | ||
|
|
6fbe25753b | ||
|
|
03f0edc5b5 | ||
|
|
5536f1e835 | ||
|
|
0de36706da | ||
|
|
17722dad6d | ||
|
|
0371599996 | ||
|
|
199649b30f | ||
|
|
4ef35488db | ||
|
|
70a91ee6ce | ||
|
|
418a27e47e | ||
|
|
61c76d02cc | ||
|
|
d98641221d | ||
|
|
8a14f87a44 | ||
|
|
02ce71734c | ||
|
|
96c2743280 | ||
|
|
7bfc34b51c | ||
|
|
40d820fd05 | ||
|
|
d69287aaf7 | ||
|
|
d475b0ba9e | ||
|
|
1638b744b7 | ||
|
|
8b19894a06 | ||
|
|
d2e0dc99de | ||
|
|
d04e40b5fd | ||
|
|
75d797b5cd | ||
|
|
ecf4891087 | ||
|
|
0e1a418678 | ||
|
|
5bef13df94 | ||
|
|
d8386121a8 | ||
|
|
000677abb6 | ||
|
|
64eb87e9b5 | ||
|
|
aa5e92bee2 | ||
|
|
0bf79dc5d6 | ||
|
|
adb2171c0a | ||
|
|
d6f8923f86 | ||
|
|
cf91ab9d5f | ||
|
|
a0fb9531db | ||
|
|
eca9353b28 | ||
|
|
a259730639 | ||
|
|
2e93d10eba | ||
|
|
70a3ceb64e | ||
|
|
b726f60afd | ||
|
|
2fa1a64999 | ||
|
|
a42b659af9 | ||
|
|
004c3230a4 | ||
|
|
2332c41510 | ||
|
|
ec3039c5a2 | ||
|
|
639d6e6071 | ||
|
|
cd518d4726 | ||
|
|
b7d9c00dff | ||
|
|
4b17575f5a | ||
|
|
5ba4bba138 | ||
|
|
b3ee5dba0f | ||
|
|
d87ff5afa9 | ||
|
|
62a24bd38f | ||
|
|
8d373c15b8 | ||
|
|
671f3e74a4 | ||
|
|
7e810233d9 | ||
|
|
4700dbd676 | ||
|
|
74e18f4317 | ||
|
|
1eea95cf18 | ||
|
|
7291b10727 | ||
|
|
6804916697 | ||
|
|
819e61bf14 | ||
|
|
b8f7e4c8ec | ||
|
|
17bcc0eed4 | ||
|
|
13003da289 | ||
|
|
9273538955 | ||
|
|
9750189def | ||
|
|
cb17ee9871 | ||
|
|
4c3b78ba9a | ||
|
|
e00b6a401b | ||
|
|
ac0ab8a7b4 | ||
|
|
0aff3941f4 | ||
|
|
27b022d4d9 | ||
|
|
e188928742 | ||
|
|
780e3c7fb7 | ||
|
|
1b5146d3ac | ||
|
|
80cf3ca6b9 | ||
|
|
2272b30a91 | ||
|
|
b5fb1cb07c | ||
|
|
4ea34a9c22 | ||
|
|
48e7de9f9e | ||
|
|
76dd2369a7 | ||
|
|
78e0cd6e77 | ||
|
|
5514a04cb4 | ||
|
|
99ca78b235 | ||
|
|
ab45db1665 | ||
|
|
bb38bcb67d | ||
|
|
6cc2912542 | ||
|
|
340b2ca624 | ||
|
|
5baa15de03 | ||
|
|
6ddca804d1 | ||
|
|
f9831a85fb | ||
|
|
7261033b7f | ||
|
|
3ad6866198 | ||
|
|
1c7d4165ab | ||
|
|
3e48b1a8ac | ||
|
|
07be100daf | ||
|
|
de9351eefb | ||
|
|
2c44b5b3a1 | ||
|
|
136f1e2fc7 | ||
|
|
ad39add55f | ||
|
|
f3c301e359 | ||
|
|
1c37a1b4d6 | ||
|
|
7222529904 | ||
|
|
f26eccd00f | ||
|
|
73375a76ac | ||
|
|
d4416d200e | ||
|
|
d1b235dd83 | ||
|
|
3ac5e0423a | ||
|
|
fc6de5f3c0 | ||
|
|
b1e475d81d | ||
|
|
4a09a4324f | ||
|
|
47f94327c5 | ||
|
|
a009ed0b6b | ||
|
|
2476a686e7 | ||
|
|
f0db93773f | ||
|
|
fabe824c8b | ||
|
|
78a077397e | ||
|
|
0c4b456aaa | ||
|
|
09185167bc | ||
|
|
5d78c3203c | ||
|
|
fa5322d3f9 | ||
|
|
d21aa5cac2 | ||
|
|
102d5c57cb | ||
|
|
ffb4de9fd9 | ||
|
|
6b3d8886e5 | ||
|
|
ddc10272a0 | ||
|
|
0b5ef00165 | ||
|
|
2a50416fc3 | ||
|
|
ce514d9f83 | ||
|
|
9b77e7fd13 | ||
|
|
abb44d3327 | ||
|
|
11eaf3d48a | ||
|
|
76c2cc2c3e | ||
|
|
0e6c8bd12e | ||
|
|
a9fb008317 | ||
|
|
e9f3a5b3e4 | ||
|
|
ebc45dff45 | ||
|
|
fc4a5ebfd3 | ||
|
|
1d7b688c55 | ||
|
|
f14a5ffbbf | ||
|
|
cada0d593c | ||
|
|
0436540791 | ||
|
|
a87ac86e18 | ||
|
|
1278b23150 | ||
|
|
02f5ea4b9d | ||
|
|
2e14e613d0 |
No files matched your search
+14
-1
@@ -7,7 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -29,11 +29,24 @@ option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
|
||||
+1
-1
@@ -135,7 +135,6 @@ set (SRCS
|
||||
Interface/IR/Passes/PhiValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
@@ -143,6 +142,7 @@ set (SRCS
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
|
||||
+10
-6
@@ -211,10 +211,12 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
|
||||
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
|
||||
@@ -629,7 +631,7 @@ namespace JSON {
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, bool Global);
|
||||
explicit AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
@@ -681,8 +683,10 @@ namespace JSON {
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, bool Global)
|
||||
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
|
||||
AppLoader::AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
@@ -754,8 +758,8 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
|
||||
+3
-2
@@ -16,10 +16,11 @@
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Default": "false",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation"
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
]
|
||||
},
|
||||
"MaxInst": {
|
||||
|
||||
@@ -20,7 +20,9 @@ namespace FEXCore::CPU {
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
: vixl::aarch64::Assembler(size ? (byte*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : reinterpret_cast<byte*>(~0ULL),
|
||||
size,
|
||||
vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
@@ -37,6 +39,13 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
SetCPUFeatures(Features);
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto CodeBuffer = GetBuffer();
|
||||
if (CodeBuffer->GetCapacity()) {
|
||||
FEXCore::Allocator::munmap(CodeBuffer->GetStartAddress<void*>(), CodeBuffer->GetCapacity());
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = Reg.IsX();
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
@@ -88,6 +88,7 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
+1
-1
@@ -421,7 +421,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(1 << 1) | // PCLMULQDQ
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(0 << 4) | // DS-CPL
|
||||
|
||||
+6
-2
@@ -44,6 +44,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
@@ -674,6 +675,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
@@ -741,6 +744,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
@@ -1011,6 +1016,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
@@ -1206,8 +1212,6 @@ namespace FEXCore::Context {
|
||||
IsMemoryShared = true;
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
LogMan::Msg::IFmt("Migrating to shared memory mode");
|
||||
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
LogMan::Throw::AFmt(Threads.size() == 1, "First MarkMemoryShared called must be before creating any threads");
|
||||
|
||||
|
||||
@@ -212,12 +212,20 @@ void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread,
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
@@ -565,10 +573,13 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
@@ -581,10 +592,8 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
|
||||
+19
-4
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
@@ -450,8 +451,13 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DestSize = 2;
|
||||
}
|
||||
else if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_128BIT);
|
||||
DestSize = 16;
|
||||
if (Options.L) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_256BIT);
|
||||
DestSize = 32;
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_128BIT);
|
||||
DestSize = 16;
|
||||
}
|
||||
}
|
||||
else if (HasNarrowingDisplacement &&
|
||||
(DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_DEF ||
|
||||
@@ -482,7 +488,14 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
|
||||
if (Options.L) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
|
||||
}
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_256BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
|
||||
}
|
||||
else if (HasNarrowingDisplacement &&
|
||||
(SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_DEF ||
|
||||
@@ -776,6 +789,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
if (Op == 0xC5) { // Two byte VEX
|
||||
pp = Byte1 & 0b11;
|
||||
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
|
||||
options.L = (Byte1 & 0b100) != 0;
|
||||
}
|
||||
else { // 0xC4 = Three byte VEX
|
||||
const uint8_t Byte2 = ReadByte();
|
||||
@@ -783,6 +797,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
map_select = Byte1 & 0b11111;
|
||||
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
options.L = (Byte2 & 0b100) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
@@ -1132,6 +1147,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1166,7 +1182,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
|
||||
@@ -49,6 +49,7 @@ private:
|
||||
struct DecodedHeader {
|
||||
uint8_t vvvv; // Encoded operand in a VEX prefix.
|
||||
bool w; // VEX.W bit.
|
||||
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
|
||||
};
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
@@ -65,6 +65,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
@@ -127,6 +128,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
SupportsPMULL_128Bit = Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
|
||||
+35
-17
@@ -894,33 +894,51 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Shift = ElementSizeBits * Op->Index;
|
||||
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
|
||||
"OpSize is too large for VExtractToGPR: {}", OpSize);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
if (SourceSize >= SSERegSize) {
|
||||
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
|
||||
|
||||
const auto GetResult = [&] {
|
||||
if (Shift >= SSEBitSize) {
|
||||
const auto NormalizedShift = Shift - SSEBitSize;
|
||||
return (Src.Upper >> NormalizedShift) & SourceMask;
|
||||
} else {
|
||||
return (Src.Lower >> Shift) & SourceMask;
|
||||
}
|
||||
};
|
||||
|
||||
const auto Result = GetResult();
|
||||
memcpy(GDP, &Result, ElementSize);
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
const uint64_t Result = (Src >> Shift) & SourceMask;
|
||||
GD = Result;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -13,22 +13,46 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
|
||||
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
const auto Scalar = Src2 & Mask;
|
||||
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
|
||||
: Offset;
|
||||
|
||||
// Now shift into place and set all bits but
|
||||
// the ones where we're going to insert our value.
|
||||
Mask <<= ScaledOffset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
Dst |= Src2 << Offset;
|
||||
|
||||
const auto Dst = [&] {
|
||||
if (InUpperLane) {
|
||||
return InterpVector256{
|
||||
.Lower = Src1.Lower,
|
||||
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
|
||||
};
|
||||
} else {
|
||||
return InterpVector256{
|
||||
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
|
||||
.Upper = Src1.Upper,
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
@@ -89,63 +113,73 @@ DEF_OP(Vector_SToF) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
@@ -165,19 +199,22 @@ DEF_OP(Vector_FToF) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
@@ -186,31 +223,31 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
|
||||
+1
-1
@@ -146,7 +146,7 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
|
||||
@@ -154,8 +154,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
@@ -245,8 +243,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VSLI, VSLI);
|
||||
REGISTER_OP(VSRI, VSRI);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
|
||||
@@ -49,7 +49,7 @@ namespace FEXCore::CPU {
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
@@ -181,8 +181,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -265,8 +263,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
|
||||
@@ -25,93 +25,139 @@ static inline void CacheLineFlush(char *Addr) {
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (IROp->Size) {
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
@@ -144,8 +190,8 @@ DEF_OP(StoreFlag) {
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -158,7 +204,8 @@ DEF_OP(LoadMem) {
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
memset(GDP, 0, 16);
|
||||
|
||||
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
|
||||
@@ -180,16 +227,15 @@ DEF_OP(LoadMem) {
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -221,41 +267,11 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
STORE_DATA(1, uint8_t)
|
||||
STORE_DATA(2, uint16_t)
|
||||
STORE_DATA(4, uint32_t)
|
||||
STORE_DATA(8, uint64_t)
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
|
||||
}
|
||||
#undef STORE_DATA
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
|
||||
@@ -19,13 +19,6 @@ $end_info$
|
||||
#include <sys/random.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(Fence) {
|
||||
|
||||
+260
-160
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <limits>
|
||||
@@ -358,23 +359,26 @@ DEF_OP(VAddP) {
|
||||
}
|
||||
|
||||
DEF_OP(VAddV) {
|
||||
auto Op = IROp->C<IR::IROp_VAddV>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VAddV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto current, auto a) { return current + a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_REDUCE_1SRC_OP(1, int8_t, Func, 0)
|
||||
DO_VECTOR_REDUCE_1SRC_OP(2, int16_t, Func, 0)
|
||||
DO_VECTOR_REDUCE_1SRC_OP(4, int32_t, Func, 0)
|
||||
DO_VECTOR_REDUCE_1SRC_OP(8, int64_t, Func, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
memcpy(GDP, Tmp, Op->Header.ElementSize);
|
||||
memcpy(GDP, Tmp, ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUMinV) {
|
||||
@@ -838,17 +842,18 @@ DEF_OP(VSMax) {
|
||||
}
|
||||
|
||||
DEF_OP(VZip) {
|
||||
auto Op = IROp->C<IR::IROp_VZip>();
|
||||
const auto Op = IROp->C<IR::IROp_VZip>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
uint8_t BaseOffset = IROp->Op == IR::OP_VZIP2 ? (Elements / 2) : 0;
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
uint8_t Elements = OpSize / ElementSize;
|
||||
const uint8_t BaseOffset = IROp->Op == IR::OP_VZIP2 ? (Elements / 2) : 0;
|
||||
Elements >>= 1;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint8_t*>(Src1);
|
||||
@@ -889,24 +894,27 @@ DEF_OP(VZip) {
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUnZip) {
|
||||
auto Op = IROp->C<IR::IROp_VUnZip>();
|
||||
const auto Op = IROp->C<IR::IROp_VUnZip>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
unsigned Start = IROp->Op == IR::OP_VUNZIP ? 0 : 1;
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
uint8_t Elements = OpSize / ElementSize;
|
||||
const unsigned Start = IROp->Op == IR::OP_VUNZIP ? 0 : 1;
|
||||
Elements >>= 1;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint8_t*>(Src1);
|
||||
@@ -947,7 +955,9 @@ DEF_OP(VUnZip) {
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
@@ -1503,16 +1513,18 @@ DEF_OP(VSShrS) {
|
||||
}
|
||||
|
||||
DEF_OP(VInsElement) {
|
||||
auto Op = IROp->C<IR::IROp_VInsElement>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VInsElement>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->DestVector);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->SrcVector);
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
// Copy src1 in to dest
|
||||
memcpy(Tmp, Src1, OpSize);
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
|
||||
auto *Src2_d = reinterpret_cast<uint8_t*>(Src2);
|
||||
@@ -1537,83 +1549,126 @@ DEF_OP(VInsElement) {
|
||||
Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx];
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
};
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VDupElement) {
|
||||
auto Op = IROp->C<IR::IROp_VDupElement>();
|
||||
const auto Op = IROp->C<IR::IROp_VDupElement>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize <= 16, "OpSize is too large for VDupElement: {}", OpSize);
|
||||
if (OpSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
const uint64_t ElementSize = Op->Header.ElementSize;
|
||||
const uint64_t ElementSizeBits = ElementSize * 8;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto Is128BitElement = ElementSizeBits == SSEBitSize;
|
||||
const auto Is256Bit = OpSize == AVXRegSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
|
||||
"OpSize is too large for VDupElement: {}", OpSize);
|
||||
|
||||
if (OpSize >= SSERegSize) {
|
||||
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
for (size_t i = 0; i < Elements; ++i) {
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * i)),
|
||||
&Src, Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
const auto GetResult = [&]() -> __uint128_t {
|
||||
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
|
||||
uint64_t Shift = ElementSizeBits * Op->Index;
|
||||
|
||||
if (Is128BitElement) {
|
||||
if (Shift == 0) {
|
||||
return Src.Lower;
|
||||
} else {
|
||||
return Src.Upper;
|
||||
}
|
||||
} else {
|
||||
// Normalize shift to act on upper uint128_t
|
||||
if (Is256Bit && Shift >= SSEBitSize) {
|
||||
Shift -= SSEBitSize;
|
||||
return (Src.Upper >> Shift) & SourceMask;
|
||||
} else {
|
||||
return (Src.Lower >> Shift) & SourceMask;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const __uint128_t Result = GetResult();
|
||||
for (size_t i = 0; i < Elements; ++i) {
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * i)),
|
||||
&Src, Op->Header.ElementSize);
|
||||
auto* Dst = static_cast<uint8_t*>(GDP) + (ElementSize * i);
|
||||
memcpy(Dst, &Result, ElementSize);
|
||||
}
|
||||
} else {
|
||||
const uint64_t Shift = ElementSizeBits * Op->Index;
|
||||
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
const uint64_t Result = (Src >> Shift) & SourceMask;
|
||||
for (size_t i = 0; i < Elements; ++i) {
|
||||
auto* Dst = static_cast<uint8_t*>(GDP) + (ElementSize * i);
|
||||
memcpy(Dst, &Result, ElementSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VExtr) {
|
||||
auto Op = IROp->C<IR::IROp_VExtr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VExtr>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto OpSizeBits = OpSize * 8;
|
||||
|
||||
const auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->VectorLower);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->VectorUpper);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Index = Op->Index;
|
||||
|
||||
uint64_t Offset = Op->Index * Op->Header.ElementSize * 8;
|
||||
__uint128_t Dst{};
|
||||
if (Offset >= (OpSize * 8)) {
|
||||
Offset -= OpSize * 8;
|
||||
Dst = Src1 >> Offset;
|
||||
if (Is256Bit) {
|
||||
const auto ByteIndex = Index * ElementSize;
|
||||
const auto IsUpperVectorZero = ByteIndex >= OpSize;
|
||||
const auto SanitizedByteIndex = IsUpperVectorZero ? ByteIndex - OpSize
|
||||
: ByteIndex;
|
||||
|
||||
const auto Vectors = IsUpperVectorZero
|
||||
?
|
||||
std::array<InterpVector256, 2>{
|
||||
*GetSrc<InterpVector256*>(Data->SSAData, Op->VectorLower),
|
||||
InterpVector256{},
|
||||
}
|
||||
:
|
||||
std::array<InterpVector256, 2>{
|
||||
*GetSrc<InterpVector256*>(Data->SSAData, Op->VectorUpper),
|
||||
*GetSrc<InterpVector256*>(Data->SSAData, Op->VectorLower),
|
||||
};
|
||||
|
||||
const auto* VectorsPtr = reinterpret_cast<const uint8_t*>(Vectors.data());
|
||||
const auto* SrcPtr = VectorsPtr + SanitizedByteIndex;
|
||||
|
||||
memcpy(GDP, SrcPtr, OpSize);
|
||||
} else {
|
||||
uint64_t Offset = Index * ElementSize * 8;
|
||||
|
||||
const auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->VectorLower);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->VectorUpper);
|
||||
|
||||
__uint128_t Dst{};
|
||||
if (Offset >= OpSizeBits) {
|
||||
Offset -= OpSizeBits;
|
||||
Dst = Src1 >> Offset;
|
||||
} else {
|
||||
Dst = (Src1 << (OpSizeBits - Offset)) | (Src2 >> Offset);
|
||||
}
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
else {
|
||||
Dst = (Src1 << (OpSize * 8 - Offset)) | (Src2 >> Offset);
|
||||
}
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSLI) {
|
||||
auto Op = IROp->C<IR::IROp_VSLI>();
|
||||
const __uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
const __uint128_t Src2 = Op->ByteShift * 8;
|
||||
|
||||
const __uint128_t Dst = Op->ByteShift >= sizeof(__uint128_t) ? 0 : Src1 << Src2;
|
||||
memcpy(GDP, &Dst, 16);
|
||||
}
|
||||
|
||||
DEF_OP(VSRI) {
|
||||
auto Op = IROp->C<IR::IROp_VSRI>();
|
||||
const __uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
const __uint128_t Src2 = Op->ByteShift * 8;
|
||||
|
||||
const __uint128_t Dst = Op->ByteShift >= sizeof(__uint128_t) ? 0 : Src1 >> Src2;
|
||||
memcpy(GDP, &Dst, 16);
|
||||
}
|
||||
|
||||
DEF_OP(VUShrI) {
|
||||
@@ -1695,205 +1750,235 @@ DEF_OP(VShlI) {
|
||||
}
|
||||
|
||||
DEF_OP(VUShrNI) {
|
||||
auto Op = IROp->C<IR::IROp_VUShrNI>();
|
||||
const auto Op = IROp->C<IR::IROp_VUShrNI>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t BitShift = Op->BitShift;
|
||||
uint8_t Tmp[16]{};
|
||||
const uint8_t BitShift = Op->BitShift;
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / (ElementSize << 1);
|
||||
const auto Func = [BitShift](auto a, auto min, auto max) {
|
||||
return BitShift >= (sizeof(a) * 8) ? 0 : a >> BitShift;
|
||||
};
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(1, uint8_t, uint16_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(2, uint16_t, uint32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, uint32_t, uint64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUShrNI2) {
|
||||
auto Op = IROp->C<IR::IROp_VUShrNI2>();
|
||||
const auto Op = IROp->C<IR::IROp_VUShrNI2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
|
||||
uint8_t BitShift = Op->BitShift;
|
||||
uint8_t Tmp[16];
|
||||
const uint8_t BitShift = Op->BitShift;
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / (ElementSize << 1);
|
||||
const auto Func = [BitShift](auto a, auto min, auto max) {
|
||||
return BitShift >= (sizeof(a) * 8) ? 0 : a >> BitShift;
|
||||
};
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(1, uint8_t, uint16_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(2, uint16_t, uint32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(4, uint32_t, uint64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSXTL) {
|
||||
auto Op = IROp->C<IR::IROp_VSXTL>();
|
||||
const auto Op = IROp->C<IR::IROp_VSXTL>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(2, int16_t, int8_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, int16_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, int32_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSXTL2) {
|
||||
auto Op = IROp->C<IR::IROp_VSXTL2>();
|
||||
const auto Op = IROp->C<IR::IROp_VSXTL2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(2, int16_t, int8_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(4, int32_t, int16_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(8, int64_t, int32_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUXTL) {
|
||||
auto Op = IROp->C<IR::IROp_VUXTL>();
|
||||
const auto Op = IROp->C<IR::IROp_VUXTL>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(2, uint16_t, uint8_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, uint32_t, uint16_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, uint64_t, uint32_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUXTL2) {
|
||||
auto Op = IROp->C<IR::IROp_VUXTL2>();
|
||||
const auto Op = IROp->C<IR::IROp_VUXTL2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSQXTN) {
|
||||
auto Op = IROp->C<IR::IROp_VSQXTN>();
|
||||
const auto Op = IROp->C<IR::IROp_VSQXTN>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / (ElementSize << 1);
|
||||
const auto Func = [](auto a, auto min, auto max) {
|
||||
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
|
||||
};
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(1, int8_t, int16_t, Func, std::numeric_limits<int8_t>::min(), std::numeric_limits<int8_t>::max())
|
||||
DO_VECTOR_1SRC_2TYPE_OP(2, int16_t, int32_t, Func, std::numeric_limits<int16_t>::min(), std::numeric_limits<int16_t>::max())
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSQXTN2) {
|
||||
auto Op = IROp->C<IR::IROp_VSQXTN2>();
|
||||
const auto Op = IROp->C<IR::IROp_VSQXTN2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / (ElementSize << 1);
|
||||
const auto Func = [](auto a, auto min, auto max) {
|
||||
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
|
||||
};
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(1, int8_t, int16_t, Func, std::numeric_limits<int8_t>::min(), std::numeric_limits<int8_t>::max())
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(2, int16_t, int32_t, Func, std::numeric_limits<int16_t>::min(), std::numeric_limits<int16_t>::max())
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSQXTUN) {
|
||||
auto Op = IROp->C<IR::IROp_VSQXTUN>();
|
||||
const auto Op = IROp->C<IR::IROp_VSQXTUN>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / (ElementSize << 1);
|
||||
const auto Func = [](auto a, auto min, auto max) {
|
||||
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
|
||||
};
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(1, uint8_t, int16_t, Func, 0, (1 << 8) - 1)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(2, uint16_t, int32_t, Func, 0, (1 << 16) - 1)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSQXTUN2) {
|
||||
auto Op = IROp->C<IR::IROp_VSQXTUN2>();
|
||||
const auto Op = IROp->C<IR::IROp_VSQXTUN2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / (ElementSize << 1);
|
||||
const auto Func = [](auto a, auto min, auto max) {
|
||||
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
|
||||
};
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(1, uint8_t, int16_t, Func, 0, (1 << 8) - 1)
|
||||
DO_VECTOR_1SRC_2TYPE_OP_TOP(2, uint16_t, int32_t, Func, 0, (1 << 16) - 1)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
@@ -1923,22 +2008,25 @@ DEF_OP(VUMul) {
|
||||
}
|
||||
|
||||
DEF_OP(VUMull) {
|
||||
auto Op = IROp->C<IR::IROp_VUMull>();
|
||||
const auto Op = IROp->C<IR::IROp_VUMull>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto b) { return a * b; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP(2, uint16_t, uint8_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(4, uint32_t, uint16_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(8, uint64_t, uint32_t, Func)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
@@ -1968,100 +2056,112 @@ DEF_OP(VSMul) {
|
||||
}
|
||||
|
||||
DEF_OP(VSMull) {
|
||||
auto Op = IROp->C<IR::IROp_VSMull>();
|
||||
const auto Op = IROp->C<IR::IROp_VSMull>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto b) { return a * b; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP(2, int16_t, int8_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(4, int32_t, int16_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(8, int64_t, int32_t, Func)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUMull2) {
|
||||
auto Op = IROp->C<IR::IROp_VUMull2>();
|
||||
const auto Op = IROp->C<IR::IROp_VUMull2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto b) { return a * b; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VSMull2) {
|
||||
auto Op = IROp->C<IR::IROp_VSMull2>();
|
||||
const auto Op = IROp->C<IR::IROp_VSMull2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func = [](auto a, auto b) { return a * b; };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, int16_t, int8_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, int32_t, int16_t, Func)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, int64_t, int32_t, Func)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL) {
|
||||
auto Op = IROp->C<IR::IROp_VUABDL>();
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func8 = [](auto a, auto b) { return std::abs((int16_t)a - (int16_t)b); };
|
||||
const auto Func16 = [](auto a, auto b) { return std::abs((int32_t)a - (int32_t)b); };
|
||||
const auto Func32 = [](auto a, auto b) { return std::abs((int64_t)a - (int64_t)b); };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP(2, uint16_t, uint8_t, Func8)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(4, uint32_t, uint16_t, Func16)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(8, uint64_t, uint32_t, Func32)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorTable);
|
||||
const auto *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorIndices);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
for (size_t i = 0; i < OpSize; ++i) {
|
||||
const uint8_t Index = Src2[i];
|
||||
|
||||
+73
-19
@@ -14,7 +14,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
@@ -1168,25 +1168,79 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V16B(), Op->Index);
|
||||
break;
|
||||
case 2:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V8H(), Op->Index);
|
||||
break;
|
||||
case 4:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V4S(), Op->Index);
|
||||
break;
|
||||
case 8:
|
||||
umov(GetReg<RA_64>(Node), GetSrc(Op->Vector.ID()).V2D(), Op->Index);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
|
||||
break;
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = Op->Header.ElementSize * 8;
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
const auto PerformMove = [&](const aarch64::VRegister& reg, int index) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov(GetReg<RA_32>(Node), reg.V16B(), index);
|
||||
break;
|
||||
case 2:
|
||||
umov(GetReg<RA_32>(Node), reg.V8H(), index);
|
||||
break;
|
||||
case 4:
|
||||
umov(GetReg<RA_32>(Node), reg.V4S(), index);
|
||||
break;
|
||||
case 8:
|
||||
umov(GetReg<RA_64>(Node), reg.V2D(), index);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
if (Offset < SSERegBitSize) {
|
||||
// Desired data lies within the lower 128-bit lane, so we
|
||||
// can treat the operation as a 128-bit operation, even
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE,
|
||||
"Host doesn't support SVE. Cannot perform 256-bit operation.");
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit,
|
||||
"Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize,
|
||||
"Trying to extract element outside bounds of register. Offset={}, Index={}",
|
||||
Offset, Op->Index);
|
||||
|
||||
// We need to use the upper 128-bit lane, so lets move it down.
|
||||
// Inverting our dedicated predicate for 128-bit operations selects
|
||||
// all of the top lanes. We can then compact those into a temporary.
|
||||
const auto CompactPred = p0;
|
||||
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
|
||||
compact(VTMP1.Z().VnD(), CompactPred, Vector.Z().VnD());
|
||||
|
||||
// Sanitize the zero-based index to work on the now-moved
|
||||
// upper half of the vector.
|
||||
const auto SanitizedIndex = [OpSize, Op] {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
return Op->Index - 16;
|
||||
case 2:
|
||||
return Op->Index - 8;
|
||||
case 4:
|
||||
return Op->Index - 4;
|
||||
case 8:
|
||||
return Op->Index - 2;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled OpSize: {}", OpSize);
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
// Move the value from the now-low-lane data.
|
||||
PerformMove(VTMP1, SanitizedIndex);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
// Size is the size of each pair element
|
||||
|
||||
@@ -19,7 +19,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -243,7 +243,6 @@ DEF_OP(InlineSyscall) {
|
||||
bool Intersects{};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
|
||||
+460
-101
@@ -10,28 +10,117 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
// This is going to be a little gross. Pls forgive me.
|
||||
// Since SVE has the whole vector length agnostic programming
|
||||
// thing going on, we can't exactly freely insert entries into
|
||||
// arbitrary locations in the vector.
|
||||
//
|
||||
// SVE *does* have INSR, however this only shifts the entire
|
||||
// vector to the left by an element size and inserts a value
|
||||
// at the beginning of the vector. Not *quite* what we need.
|
||||
// (though INSR *is* very useful for other things).
|
||||
//
|
||||
// The idea is (in the case of the upper lane), move the upper
|
||||
// lane down, insert into it and recombine with the lower lane.
|
||||
//
|
||||
// In the case of the lower lane, insert and then recombine with
|
||||
// the upper lane.
|
||||
|
||||
if (InUpperLane) {
|
||||
// Move the upper lane down for the insertion.
|
||||
const auto CompactPred = p0;
|
||||
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
|
||||
compact(VTMP1.Z().VnD(), CompactPred, DestVector.Z().VnD());
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
|
||||
// Put data in place for destructive SPLICE below.
|
||||
mov(Dst.Z().VnD(), DestVector.Z().VnD());
|
||||
|
||||
// Inserts the GPR value into the given V register.
|
||||
// Also automatically adjusts the index in the case of using the
|
||||
// moved upper lane.
|
||||
const auto Insert = [&](const aarch64::VRegister& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
ins(reg.V16B(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 2:
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
ins(reg.V8H(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 4:
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
ins(reg.V4S(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 8:
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
ins(reg.V2D(), index, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
Insert(VTMP1, DestIdx);
|
||||
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), VTMP1.Z().VnD());
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), DestVector.Z().VnD());
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
} else {
|
||||
mov(Dst, DestVector);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
ins(Dst.V16B(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(Dst.V8H(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(Dst.V4S(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(Dst.V2D(), DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,8 +146,11 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
@@ -76,6 +168,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -96,116 +192,379 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
scvtf(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
scvtf(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
scvtf(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
scvtf(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
fcvtzs(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
fcvtzs(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
fcvtzs(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
fcvtzs(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
fcvtzs(Dst.Z().VnH(), Mask, Dst.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
fcvtzs(Dst.Z().VnS(), Mask, Dst.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
fcvtzs(Dst.Z().VnD(), Mask, Dst.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.V8H(), Vector.V8H());
|
||||
fcvtzs(Dst.V8H(), Dst.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.V4S(), Vector.V4S());
|
||||
fcvtzs(Dst.V4S(), Dst.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.V2D(), Vector.V2D());
|
||||
fcvtzs(Dst.V2D(), Dst.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// Curiously, FCVTLT and FCVTNT have no bottom variants,
|
||||
// and also interesting is that FCVTLT will iterate the
|
||||
// source vector by accessing each odd element and storing
|
||||
// them consecutively in the destination.
|
||||
//
|
||||
// FCVTNT is somewhat like the opposite. It will read each
|
||||
// consecutive element, but store each result into every odd
|
||||
// element in the destination vector.
|
||||
//
|
||||
// We need to undo the behavior of FCVTNT with UZP2. In the case
|
||||
// of FCVTLT, we instead need to set the vector up with ZIP1, so
|
||||
// that the elements will be processed correctly.
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(Dst.Z().VnH(), Vector.Z().VnH(), Vector.Z().VnH());
|
||||
fcvtlt(Dst.Z().VnS(), Mask, Dst.Z().VnH());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(Dst.Z().VnS(), Vector.Z().VnS(), Vector.Z().VnS());
|
||||
fcvtlt(Dst.Z().VnD(), Mask, Dst.Z().VnS());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(Dst.Z().VnH(), Mask, Vector.Z().VnS());
|
||||
uzp2(Dst.Z().VnH(), Dst.Z().VnH(), Dst.Z().VnH());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(Dst.Z().VnS(), Mask, Vector.Z().VnD());
|
||||
uzp2(Dst.Z().VnS(), Dst.Z().VnS(), Dst.Z().VnS());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
} else {
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
fcvtl(Dst.V4S(), Vector.V4H());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(Dst.V2D(), Vector.V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtn(Dst.V4H(), Vector.V4S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(Dst.V2S(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintn(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintn(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintn(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintm(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintm(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintm(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintp(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintp(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintp(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintz(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintz(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintz(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
}
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintn(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintn(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintm(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintm(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintp(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintp(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintz(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintz(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
@@ -18,7 +18,7 @@ DEF_OP(AESImc) {
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -27,7 +27,7 @@ DEF_OP(AESEnc) {
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -35,7 +35,7 @@ DEF_OP(AESEncLast) {
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -44,7 +44,7 @@ DEF_OP(AESDec) {
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
@@ -52,7 +52,7 @@ DEF_OP(AESDecLast) {
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label PastConstant;
|
||||
@@ -69,9 +69,8 @@ DEF_OP(AESKeyGenAssist) {
|
||||
if (Op->RCON) {
|
||||
tbl(VTMP1.V16B(), VTMP1.V16B(), VTMP3.V16B());
|
||||
|
||||
LoadConstant(TMP1.W(), Op->RCON);
|
||||
ins(VTMP2.V4S(), 1, TMP1.W());
|
||||
ins(VTMP2.V4S(), 3, TMP1.W());
|
||||
LoadConstant(TMP1, static_cast<uint64_t>(Op->RCON) << 32);
|
||||
dup(VTMP2.V2D(), TMP1);
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), VTMP2.V16B());
|
||||
}
|
||||
else {
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
|
||||
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -79,7 +80,7 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -476,7 +477,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
@@ -731,6 +732,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
+13
-6
@@ -118,6 +118,17 @@ private:
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]] SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
|
||||
aarch64::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
@@ -202,7 +213,7 @@ private:
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
@@ -214,7 +225,7 @@ private:
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
@@ -334,8 +345,6 @@ private:
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -419,8 +428,6 @@ private:
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
|
||||
+723
-356
File diff suppressed because it is too large.
Load diff
+14
-11
@@ -11,7 +11,7 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
@@ -41,16 +41,19 @@ DEF_OP(Break) {
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Op->Reason.Signal,
|
||||
.TrapNo = Op->Reason.TrapNumber,
|
||||
.si_code = Op->Reason.si_code,
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(x1, Constant);
|
||||
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData)));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
|
||||
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
|
||||
+1382
-574
File diff suppressed because it is too large.
Load diff
+44
-11
@@ -1143,26 +1143,59 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
|
||||
const auto Is256Bit = Offset >= SSEBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrb(GetDst<RA_32>(Node), xmm15, Op->Index - 16);
|
||||
} else {
|
||||
pextrb(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrw(GetDst<RA_32>(Node), xmm15, Op->Index - 8);
|
||||
} else {
|
||||
pextrw(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrd(GetDst<RA_32>(Node), xmm15, Op->Index - 4);
|
||||
} else {
|
||||
pextrd(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrq(GetDst<RA_64>(Node), xmm15, Op->Index - 2);
|
||||
} else {
|
||||
pextrq(GetDst<RA_64>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+224
-74
@@ -17,27 +17,76 @@ namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
if (InUpperLane && !Is256Bit) {
|
||||
LOGMAN_MSG_A_FMT("Attempt to access upper 128-bit lane in 128-bit operation! Offset={}",
|
||||
Offset);
|
||||
return;
|
||||
}
|
||||
|
||||
if (Is256Bit) {
|
||||
vmovapd(ToYMM(Dst), ToYMM(DestVector));
|
||||
} else {
|
||||
vmovapd(Dst, DestVector);
|
||||
}
|
||||
|
||||
const auto Insert = [&](const Xbyak::Xmm& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
pinsrb(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
pinsrw(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
pinsrd(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
pinsrq(reg, GetSrc<RA_64>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
vextracti128(xmm15, ToYMM(Dst), 1);
|
||||
Insert(xmm15, DestIdx);
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,8 +112,10 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
@@ -83,6 +134,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -104,99 +159,194 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtdq2ps(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtdq2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
// There is no vector form of this instruction until AVX512VL + AVX512DQ (vcvtqq2pd)
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
pextrq(rax, Vector, 1);
|
||||
pextrq(rcx, Vector, 0);
|
||||
cvtsi2sd(Dst, rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
vmovlhps(Dst, Dst, xmm15);
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
|
||||
pextrq(rax, xmm15, 1);
|
||||
pextrq(rcx, xmm15, 0);
|
||||
cvtsi2sd(xmm15, rcx);
|
||||
cvtsi2sd(xmm14, rax);
|
||||
movlhps(xmm15, xmm14);
|
||||
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvttps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvttpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvtpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtps2pd(ToYMM(Dst), Vector);
|
||||
} else {
|
||||
vcvtps2pd(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtpd2ps(Dst, ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t RoundMode{};
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
RoundMode = 0b0000'0'0'00;
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'01;
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'10;
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
RoundMode = 0b0000'0'0'11;
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
RoundMode = 0b0000'0'1'00;
|
||||
break;
|
||||
}
|
||||
const uint8_t RoundMode = [Op] {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
return 0b0000'0'0'00;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
return 0b0000'0'0'01;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
return 0b0000'0'0'10;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
return 0b0000'0'0'11;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
return 0b0000'0'1'00;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled rounding mode");
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundps(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundps(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundpd(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundpd(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -370,7 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
|
||||
{
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -582,6 +583,8 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("x86::CompileCode");
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
|
||||
@@ -337,6 +337,8 @@ private:
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
@@ -345,8 +347,6 @@ private:
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
@@ -430,8 +430,6 @@ private:
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
|
||||
+334
-176
@@ -21,130 +21,266 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(GetDst<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(GetDst<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(GetDst<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid GPR load of size 16");
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(GetDst(Node), dword [STATE + Op->Offset]);
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(GetDst(Node), qword [STATE + Op->Offset]);
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16:
|
||||
LogMan::Msg::DFmt("Invalid store size of 16");
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid store size of 16");
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrb(byte [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrw(word [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovd(dword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovq(qword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetSrc<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
movzx(GetSrc<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
mov(GetSrc<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
mov(GetSrc<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetSrc(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -153,21 +289,21 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 2:
|
||||
movzx(GetDst<RA_32>(Node), word [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), word [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 4:
|
||||
mov(GetDst<RA_32>(Node), dword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_32>(Node), dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_64>(Node), qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -186,53 +322,63 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(eax, byte [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, byte [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(eax, word [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, word [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [rax + index * Op->Stride]);
|
||||
vmovd(Dst, dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
|
||||
vmovq(Dst, qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pinsrb(GetDst(Node), byte [STATE + rax], 0);
|
||||
pinsrb(Dst, byte [STATE + rax], 0);
|
||||
break;
|
||||
case 2:
|
||||
pinsrw(GetDst(Node), word [STATE + rax], 0);
|
||||
pinsrw(Dst, word [STATE + rax], 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [STATE + rax]);
|
||||
vmovd(Dst, dword [STATE + rax]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [STATE + rax]);
|
||||
vmovq(Dst, qword [STATE + rax]);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + rax]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + rax]);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + rax]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + rax]);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
vmovups(ToYMM(Dst), yword [STATE + rax]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -245,12 +391,13 @@ DEF_OP(LoadContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
size_t size = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto value = GetSrc<RA_64>(Op->Value.ID());
|
||||
const auto Value = GetSrc<RA_64>(Op->Value.ID());
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -258,10 +405,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
if (!(OpSize == 1 || OpSize == 2 || OpSize == 4 || OpSize == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
}
|
||||
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
mov(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -270,57 +417,64 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(xword [STATE + rax], value);
|
||||
else
|
||||
movups(xword [STATE + rax], value);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(xword [STATE + rax], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + rax], Value);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [STATE + rax], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -376,7 +530,7 @@ DEF_OP(SpillRegister) {
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovaps(yword [rsp + SlotOffset], ToYMM(Src));
|
||||
vmovups(yword [rsp + SlotOffset], ToYMM(Src));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -420,19 +574,19 @@ DEF_OP(FillRegister) {
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(Dst, dword [rsp + SlotOffset]);
|
||||
vmovss(Dst, dword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(Dst, qword [rsp + SlotOffset]);
|
||||
vmovsd(Dst, qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(Dst, xword [rsp + SlotOffset]);
|
||||
vmovaps(Dst, xword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovaps(ToYMM(Dst), yword [rsp + SlotOffset]);
|
||||
vmovups(ToYMM(Dst), yword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -482,130 +636,136 @@ Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
const auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx (Dst, byte [MemPtr]);
|
||||
movzx(Dst, byte [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx (Dst, word [MemPtr]);
|
||||
movzx(Dst, word [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(Dst.cvt32(), dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
auto Dst = GetDst(Node);
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(eax, byte [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(eax, word [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(Dst, dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
else
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, GetDst(Node));
|
||||
}
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
vmovups(Dst, xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 2:
|
||||
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 4:
|
||||
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 8:
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrb(byte [MemPtr], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrw(word [MemPtr], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovd(dword [MemPtr], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovq(qword [MemPtr], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
vmovups(xword [MemPtr], Value);
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [MemPtr], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
@@ -636,8 +796,8 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP(LOADREGISTER, Unhandled); // SRA specific, not supported on this backend
|
||||
REGISTER_OP(STOREREGISTER, Unhandled);
|
||||
REGISTER_OP(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
@@ -648,8 +808,6 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
|
||||
@@ -50,11 +50,19 @@ DEF_OP(Break) {
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)], Op->Reason.Signal);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], Op->Reason.TrapNumber);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], Op->Reason.ErrorRegister);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], Op->Reason.si_code);
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Op->Reason.Signal,
|
||||
.TrapNo = Op->Reason.TrapNumber,
|
||||
.si_code = Op->Reason.si_code,
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
mov(TMP1, Constant);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData)], TMP1);
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
|
||||
+2150
-737
File diff suppressed because it is too large.
Load diff
+9
-8
@@ -8,6 +8,7 @@
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -17,7 +18,7 @@ namespace Context {
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
struct LookupCacheEntry {
|
||||
struct LookupCacheEntry {
|
||||
uintptr_t HostCode;
|
||||
uintptr_t GuestCode;
|
||||
};
|
||||
@@ -54,7 +55,7 @@ public:
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
@@ -62,7 +63,7 @@ public:
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
@@ -73,7 +74,7 @@ public:
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -88,7 +89,7 @@ public:
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
@@ -158,7 +159,7 @@ public:
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
@@ -167,7 +168,7 @@ public:
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
@@ -239,7 +240,7 @@ private:
|
||||
|
||||
|
||||
std::map<BlockLinkTag, std::function<void()>> BlockLinks;
|
||||
std::map<uint64_t, uint64_t> BlockList;
|
||||
tsl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
|
||||
+845
-362
File diff suppressed because it is too large.
Load diff
+37
-3
@@ -76,10 +76,10 @@ public:
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::Context *CTX{};
|
||||
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
@@ -278,6 +278,12 @@ public:
|
||||
void NOTOp(OpcodeArgs);
|
||||
void XADDOp(OpcodeArgs);
|
||||
void PopcountOp(OpcodeArgs);
|
||||
void DAAOp(OpcodeArgs);
|
||||
void DASOp(OpcodeArgs);
|
||||
void AAAOp(OpcodeArgs);
|
||||
void AASOp(OpcodeArgs);
|
||||
void AAMOp(OpcodeArgs);
|
||||
void AADOp(OpcodeArgs);
|
||||
void XLATOp(OpcodeArgs);
|
||||
template<bool Reseed>
|
||||
void RDRANDOp(OpcodeArgs);
|
||||
@@ -292,6 +298,8 @@ public:
|
||||
void WriteSegmentReg(OpcodeArgs);
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
|
||||
// SSE
|
||||
void MOVAPSOp(OpcodeArgs);
|
||||
void MOVUPSOp(OpcodeArgs);
|
||||
@@ -398,6 +406,26 @@ public:
|
||||
// ADX Ops
|
||||
void ADXOp(OpcodeArgs);
|
||||
|
||||
// AVX Ops
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
|
||||
void VANDNOp(OpcodeArgs);
|
||||
|
||||
void VMOVAPS_VMOVAPD_Op(OpcodeArgs);
|
||||
void VMOVUPS_VMOVUPD_Op(OpcodeArgs);
|
||||
|
||||
void VMOVHPOp(OpcodeArgs);
|
||||
void VMOVLPOp(OpcodeArgs);
|
||||
|
||||
void VMOVDDUPOp(OpcodeArgs);
|
||||
void VMOVSHDUPOp(OpcodeArgs);
|
||||
void VMOVSLDUPOp(OpcodeArgs);
|
||||
|
||||
void VMOVVectorNTOp(OpcodeArgs);
|
||||
|
||||
void VZEROOp(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
template<size_t width>
|
||||
void FLD(OpcodeArgs);
|
||||
@@ -515,7 +543,7 @@ public:
|
||||
void X87FRSTORF64(OpcodeArgs);
|
||||
void X87FXAMF64(OpcodeArgs);
|
||||
void X87LDENVF64(OpcodeArgs);
|
||||
|
||||
|
||||
template<size_t width, bool Integer, FCOMIFlags whichflags, bool poptwice>
|
||||
void FCOMIF64(OpcodeArgs);
|
||||
|
||||
@@ -646,6 +674,7 @@ private:
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
@@ -657,6 +686,11 @@ private:
|
||||
// Non-temporal streaming
|
||||
ACCESS_STREAM,
|
||||
};
|
||||
OrderedNode *LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0);
|
||||
OrderedNode *LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, OrderedNode *const Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, OrderedNode *const Src);
|
||||
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
@@ -221,7 +221,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *XMM0 = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
|
||||
@@ -769,7 +769,11 @@ void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, Orde
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
|
||||
auto OpSize = SrcSize * 8;
|
||||
if (OpSize < Shift) {
|
||||
Shift &= (OpSize - 1);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, OpSize - Shift, Src1));
|
||||
}
|
||||
|
||||
// PF
|
||||
@@ -934,6 +938,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
|
||||
// If shift == 0, don't update flags
|
||||
@@ -963,7 +968,9 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode
|
||||
// OF
|
||||
{
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF);
|
||||
@@ -977,8 +984,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
|
||||
auto NewCF = _Bfe(1, OpSize - 1, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -989,8 +995,10 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Res), NewCF));
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1000,17 +1008,22 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, Ord
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, 0, Res);
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
}
|
||||
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is the top two MSBs XOR'd together
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1)));
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+230
-59
@@ -33,11 +33,46 @@ void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVVectorNTOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1, true, false, MemoryAccessType::ACCESS_STREAM);
|
||||
|
||||
// TODO: When stores and loads gain the ability to explicitly express
|
||||
// whether a vector extension or an insert is desirable, ensure
|
||||
// the 128-bit case here is a zero extend on store if the destination
|
||||
// is a register.
|
||||
|
||||
StoreResult(FPRClass, Op, Src, 1, MemoryAccessType::ACCESS_STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVAPSOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVAPS_VMOVAPD_Op(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto Is128BitDest = GetDstSize(Op) == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
if (Op->Dest.IsGPR() && Is128BitDest) {
|
||||
// Perform 32 byte store to clear the upper lane.
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 32, -1);
|
||||
} else {
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVUPS_VMOVUPD_Op(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1);
|
||||
const auto Is128BitDest = GetDstSize(Op) == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
if (Op->Dest.IsGPR() && Is128BitDest) {
|
||||
// Perform 32 byte store to clear the upper lane.
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 32, 1);
|
||||
} else {
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVUPSOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1);
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
@@ -68,6 +103,19 @@ void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVHPOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 16);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 8, 1, 0, Src1, Src2);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
} else {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 16);
|
||||
OrderedNode *Result = _VInsElement(16, 8, 0, 1, Src, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
if (Op->Dest.IsGPR()) {
|
||||
@@ -78,9 +126,10 @@ void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 16, 16);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16);
|
||||
auto DstSize = GetDstSize(Op);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
auto Result = _VInsElement(16, 8, 0, 0, Dest, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 16);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -88,24 +137,62 @@ void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVLPOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 16);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 8, 0, 0, Src1, Src2);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
} else {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 8, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSHDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 4, 3, 3, Src, Src);
|
||||
Result = _VInsElement(16, 4, 2, 3, Result, Src);
|
||||
Result = _VInsElement(16, 4, 1, 1, Result, Src);
|
||||
OrderedNode *Result = _VInsElement(16, 4, 2, 3, Src, Src);
|
||||
Result = _VInsElement(16, 4, 0, 1, Result, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVSHDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
OrderedNode *Result = _VInsElement(SrcSize, 4, 2, 3, Src, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 0, 1, Result, Src);
|
||||
if (Is256Bit) {
|
||||
Result = _VInsElement(SrcSize, 4, 4, 5, Result, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 6, 7, Result, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSLDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8);
|
||||
OrderedNode *Result = _VInsElement(16, 4, 3, 2, Src, Src);
|
||||
Result = _VInsElement(16, 4, 2, 2, Result, Src);
|
||||
Result = _VInsElement(16, 4, 1, 0, Result, Src);
|
||||
Result = _VInsElement(16, 4, 0, 0, Result, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVSLDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
OrderedNode *Result = _VInsElement(SrcSize, 4, 3, 2, Src, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 1, 0, Result, Src);
|
||||
if (Is256Bit) {
|
||||
Result = _VInsElement(SrcSize, 4, 5, 4, Result, Src);
|
||||
Result = _VInsElement(SrcSize, 4, 7, 6, Result, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 32, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) {
|
||||
// MOVSS xmm1, xmm2
|
||||
@@ -301,6 +388,48 @@ void OpDispatchBuilder::VectorALUOp<IR::OP_VUQSUB, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUOp<IR::OP_VUQSUB, 2>(OpcodeArgs);
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorALUOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _VAdd(Size, ElementSize, Src1, Src2);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
if (Is128Bit) {
|
||||
// 128-bit variants need to zero the upper lane.
|
||||
Result = _VMov(Size, ALUOp);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VADD, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VAND, 16>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VOR, 16>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorALUOp<IR::OP_VXOR, 16>(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorALUROp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
@@ -321,8 +450,9 @@ void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 8>(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -332,9 +462,9 @@ void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
|
||||
if (Size != ElementSize) {
|
||||
if (DstSize != ElementSize) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsElement(Size, ElementSize, 0, 0, Dest, Result);
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -368,11 +498,12 @@ void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 8>(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
auto DstSize = GetDstSize(Op);
|
||||
if constexpr (Scalar) {
|
||||
Size = ElementSize;
|
||||
}
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _VFSqrt(Size, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
@@ -380,7 +511,7 @@ void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
auto Result = _VInsElement(GetSrcSize(Op), ElementSize, 0, 0, Dest, ALUOp);
|
||||
auto Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else {
|
||||
@@ -441,14 +572,8 @@ void OpDispatchBuilder::MOVQOp(OpcodeArgs) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
const auto fprLowOffset = CTX->HostFeatures.SupportsAVX ? offsetof(Core::CPUState, xmm.avx.data[gprIndex][0])
|
||||
: offsetof(Core::CPUState, xmm.sse.data[gprIndex][0]);
|
||||
const auto fprHighOffset = CTX->HostFeatures.SupportsAVX ? offsetof(Core::CPUState, xmm.avx.data[gprIndex][1])
|
||||
: offsetof(Core::CPUState, xmm.sse.data[gprIndex][1]);
|
||||
|
||||
_StoreContext(8, FPRClass, Src, fprLowOffset);
|
||||
auto Const = _Constant(0);
|
||||
_StoreContext(8, GPRClass, Const, fprHighOffset);
|
||||
auto Reg = _VMov(16, Src);
|
||||
StoreXMMRegister(gprIndex, Reg);
|
||||
}
|
||||
else {
|
||||
// This is simple, just store the result
|
||||
@@ -659,6 +784,22 @@ void OpDispatchBuilder::ANDNOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VANDNOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
|
||||
Src1 = _VNot(Size, Size, Src1);
|
||||
OrderedNode *Dest = _VAnd(Size, Size, Src1, Src2);
|
||||
if (Is128Bit) {
|
||||
Dest = _VMov(16, Dest);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PINSROp(OpcodeArgs) {
|
||||
auto Size = GetDstSize(Op);
|
||||
@@ -926,7 +1067,11 @@ void OpDispatchBuilder::PSRLDQ(OpcodeArgs) {
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
|
||||
auto Result = _VSRI(Size, 16, Dest, Shift);
|
||||
OrderedNode *Result = _VectorZero(Size);
|
||||
if (Shift < Size) {
|
||||
Result = _VExtr(Size, 1, Result, Dest, Shift);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -938,7 +1083,10 @@ void OpDispatchBuilder::PSLLDQ(OpcodeArgs) {
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
|
||||
auto Result = _VSLI(Size, 16, Dest, Shift);
|
||||
OrderedNode *Result = _VectorZero(Size);
|
||||
if (Shift < Size) {
|
||||
Result = _VExtr(Size, 1, Dest, Result, Size - Shift);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -982,6 +1130,23 @@ void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMOVDDUPOp(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto IsSrcGPR = Op->Src[0].IsGPR();
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto MemSize = Is256Bit ? 32 : 8;
|
||||
|
||||
OrderedNode *Src = IsSrcGPR ? LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1)
|
||||
: LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], MemSize, Op->Flags, -1);
|
||||
|
||||
OrderedNode *Res = _VInsElement(SrcSize, 8, 1, 0, Src, Src);
|
||||
if (Is256Bit) {
|
||||
Res = _VInsElement(SrcSize, 8, 3, 2, Res, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Res, 32, -1);
|
||||
}
|
||||
|
||||
template<size_t DstElementSize>
|
||||
void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
@@ -1087,13 +1252,16 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
Src = _Float_FToF(DstElementSize, SrcElementSize, Src);
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
auto Result = _VInsElement(DstSize, DstElementSize, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
@@ -1103,8 +1271,9 @@ void OpDispatchBuilder::Scalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
size_t Size = GetDstSize(Op);
|
||||
|
||||
if constexpr (DstElementSize > SrcElementSize) {
|
||||
Src = _Vector_FToF(Size, SrcElementSize << 1, Src, SrcElementSize);
|
||||
@@ -1181,13 +1350,12 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
// Until we get correct PHI nodes this is required to be a loop unroll
|
||||
const auto GPRSize = CTX->GetGPRSize();
|
||||
const auto Size = uint32_t{GetSrcSize(Op)} * 8;
|
||||
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
OrderedNode *MemDest = _LoadContext(GPRSize, GPRClass, GPROffset(X86State::REG_RDI));
|
||||
auto MemDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
const size_t NumElements = Size / 64;
|
||||
for (size_t Element = 0; Element < NumElements; ++Element) {
|
||||
@@ -1398,16 +1566,9 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
const auto GetXMMOffset = [this](size_t i) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
return offsetof(Core::CPUState, xmm.avx.data[i]);
|
||||
} else {
|
||||
return offsetof(Core::CPUState, xmm.sse.data[i]);
|
||||
}
|
||||
};
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = _LoadContext(16, FPRClass, GetXMMOffset(i));
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
@@ -1455,18 +1616,11 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
const auto GetXMMOffset = [this](size_t i) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
return offsetof(Core::CPUState, xmm.avx.data[i]);
|
||||
} else {
|
||||
return offsetof(Core::CPUState, xmm.sse.data[i]);
|
||||
}
|
||||
};
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, XMMReg, GetXMMOffset(i));
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1563,6 +1717,9 @@ void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
|
||||
static_assert(ElementSize == sizeof(uint32_t),
|
||||
"Currently only handles 32-bit -> 64-bit");
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
@@ -1579,17 +1736,8 @@ void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
OrderedNode* Srcs1[2]{};
|
||||
OrderedNode* Srcs2[2]{};
|
||||
|
||||
Srcs1[0] = _VExtr(Size, ElementSize, Src1, Src1, 0);
|
||||
Srcs1[1] = _VExtr(Size, ElementSize, Src1, Src1, 2);
|
||||
|
||||
Srcs2[0] = _VExtr(Size, ElementSize, Src2, Src2, 0);
|
||||
Srcs2[1] = _VExtr(Size, ElementSize, Src2, Src2, 2);
|
||||
|
||||
Src1 = _VInsElement(Size, ElementSize, 1, 0, Srcs1[0], Srcs1[1]);
|
||||
Src2 = _VInsElement(Size, ElementSize, 1, 0, Srcs2[0], Srcs2[1]);
|
||||
Src1 = _VInsElement(Size, ElementSize, 1, 2, Src1, Src1);
|
||||
Src2 = _VInsElement(Size, ElementSize, 1, 2, Src2, Src2);
|
||||
|
||||
if constexpr (Signed) {
|
||||
Res = _VSMull(Size, ElementSize, Src1, Src2);
|
||||
@@ -1613,11 +1761,9 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) {
|
||||
// This instruction is a bit special in that if the source is MMX then it zexts to 128bit
|
||||
if constexpr (ToXMM) {
|
||||
const auto Index = Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0;
|
||||
const auto Offset = CTX->HostFeatures.SupportsAVX ? offsetof(FEXCore::Core::CPUState, xmm.avx.data[Index][0])
|
||||
: offsetof(FEXCore::Core::CPUState, xmm.sse.data[Index][0]);
|
||||
|
||||
Src = _VMov(16, Src);
|
||||
_StoreContext(16, FPRClass, Src, Offset);
|
||||
StoreXMMRegister(Index, Src);
|
||||
}
|
||||
else {
|
||||
// This is simple, just store the result
|
||||
@@ -2377,7 +2523,8 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
// The mask is hardcoded to be xmm0 in this instruction
|
||||
OrderedNode *Mask = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
auto Mask = LoadXMMRegister(0);
|
||||
|
||||
// Each element is selected by the high bit of that element size
|
||||
// Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest;
|
||||
//
|
||||
@@ -2612,4 +2759,28 @@ void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VZEROOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto IsVZEROALL = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
if (IsVZEROALL) {
|
||||
// NOTE: Despite the name being VZEROALL, this will still only ever
|
||||
// zero out up to the first 16 registers (even on AVX-512, where we have 32 registers)
|
||||
|
||||
OrderedNode* ZeroVector = _VectorZero(DstSize);
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
StoreXMMRegister(i, ZeroVector);
|
||||
}
|
||||
} else {
|
||||
// Likewise, VZEROUPPER will only ever zero only up to the first 16 registers
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
OrderedNode* Reg = LoadXMMRegister(i);
|
||||
OrderedNode* Dst = _VMov(16, Reg);
|
||||
StoreXMMRegister(i, Dst);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -782,6 +782,12 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -809,7 +815,8 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -854,6 +861,9 @@ void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
|
||||
auto sin = _F80SIN(a);
|
||||
auto cos = _F80COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -900,6 +910,9 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
OrderedNode *data = _VCastFromGPR(16, 8, low);
|
||||
data = _VInsGPR(16, 8, 1, data, high);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
@@ -778,6 +778,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -804,7 +810,8 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -831,6 +838,9 @@ void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
|
||||
auto sin = _F64SIN(a);
|
||||
auto cos = _F64COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -871,6 +881,9 @@ void OpDispatchBuilder::X87TANF64(OpcodeArgs) {
|
||||
|
||||
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(one, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
@@ -266,10 +266,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x17, 1, X86InstInfo{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1E, 1, X86InstInfo{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1F, 1, X86InstInfo{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
{0x48, 8, X86InstInfo{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
@@ -283,8 +283,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -76,7 +76,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -85,7 +85,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -94,7 +94,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -295,7 +295,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
|
||||
{0x20, 4, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
{0x24, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"CVTSI2SS", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
{0x2B, 1, X86InstInfo{"MOVNTSS", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x2C, 1, X86InstInfo{"CVTTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
{0x2D, 1, X86InstInfo{"CVTSS2SI", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR, 0, nullptr}},
|
||||
@@ -305,18 +305,18 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 1, X86InstInfo{"RSQRTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x53, 1, X86InstInfo{"RCPSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 1, X86InstInfo{"RSQRTSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x53, 1, X86InstInfo{"RCPSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x54, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSS2SD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSS2SD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5B, 1, X86InstInfo{"CVTTPS2DQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x60, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x68, 7, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -383,16 +383,16 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x40, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x50, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x51, 1, X86InstInfo{"SQRTSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x52, 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x58, 1, X86InstInfo{"ADDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x59, 1, X86InstInfo{"MULSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5A, 1, X86InstInfo{"CVTSD2SS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5B, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5C, 1, X86InstInfo{"SUBSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5D, 1, X86InstInfo{"MINSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5E, 1, X86InstInfo{"DIVSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x5F, 1, X86InstInfo{"MAXSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x60, 16, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
|
||||
+51
-51
@@ -17,23 +17,23 @@ void InitializeVEXTables() {
|
||||
static constexpr U16U8InfoStruct VEXTable[] = {
|
||||
// Map 0 (Reserved)
|
||||
// VEX Map 1
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x10), 1, X86InstInfo{"VMODUPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x10), 1, X86InstInfo{"VMOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x10), 1, X86InstInfo{"VMOVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x10), 1, X86InstInfo{"VMOVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x11), 1, X86InstInfo{"VMOVUPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x11), 1, X86InstInfo{"VMODUPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x11), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x11), 1, X86InstInfo{"VMOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x11), 1, X86InstInfo{"VMOVSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x11), 1, X86InstInfo{"VMOVSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x13), 1, X86InstInfo{"VMOVLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x13), 1, X86InstInfo{"VMOVLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x13), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x13), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x14), 1, X86InstInfo{"VUNPCKLPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x14), 1, X86InstInfo{"VUNPCKLPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -41,12 +41,12 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b00, 0x15), 1, X86InstInfo{"VUNPCKHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x15), 1, X86InstInfo{"VUNPCKHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x16), 1, X86InstInfo{"VMOVHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x16), 1, X86InstInfo{"VMOVHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x16), 1, X86InstInfo{"VMOVSHDUP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x16), 1, X86InstInfo{"VMOVHPS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x16), 1, X86InstInfo{"VMOVHPD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x16), 1, X86InstInfo{"VMOVSHDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x17), 1, X86InstInfo{"VMOVHPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x17), 1, X86InstInfo{"VMOVHPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x17), 1, X86InstInfo{"VMOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x17), 1, X86InstInfo{"VMOVHPD", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x50), 1, X86InstInfo{"VMOVMSKPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x50), 1, X86InstInfo{"VMOVMSKPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -62,17 +62,17 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b00, 0x53), 1, X86InstInfo{"VRCPPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x53), 1, X86InstInfo{"VRCPSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x54), 1, X86InstInfo{"VANDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x54), 1, X86InstInfo{"VANDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x54), 1, X86InstInfo{"VANDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x54), 1, X86InstInfo{"VANDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x55), 1, X86InstInfo{"VANDNPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x55), 1, X86InstInfo{"VANDNPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x55), 1, X86InstInfo{"VANDNPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x55), 1, X86InstInfo{"VANDNPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x56), 1, X86InstInfo{"VORPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x56), 1, X86InstInfo{"VORPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x56), 1, X86InstInfo{"VORPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x56), 1, X86InstInfo{"VORPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x57), 1, X86InstInfo{"VXORPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x57), 1, X86InstInfo{"VDORPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x57), 1, X86InstInfo{"VXORPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x57), 1, X86InstInfo{"VXORPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x60), 1, X86InstInfo{"VPUNPCKLBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x61), 1, X86InstInfo{"VPUNPCKLWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -95,7 +95,7 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x75), 1, X86InstInfo{"VPCMPEQW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x76), 1, X86InstInfo{"VPCMPEQD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x77), 1, X86InstInfo{"VZERO*", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x77), 1, X86InstInfo{"VZERO*", TYPE_INST, GenFlagsDstSize(SIZE_128BIT), 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xC2), 1, X86InstInfo{"VCMPccPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xC2), 1, X86InstInfo{"VCMPccPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -112,17 +112,17 @@ void InitializeVEXTables() {
|
||||
// This table doesn't state which VEX.pp is for which instruction
|
||||
// XXX: Confirm all the above encoding opcodes
|
||||
|
||||
{OPD(1, 0b00, 0x28), 1, X86InstInfo{"VMOVAPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x28), 1, X86InstInfo{"VMOVAPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x28), 1, X86InstInfo{"VMOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x28), 1, X86InstInfo{"VMOVAPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x29), 1, X86InstInfo{"VMOVAPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x29), 1, X86InstInfo{"VMOVAPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x29), 1, X86InstInfo{"VMOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x29), 1, X86InstInfo{"VMOVAPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, X86InstInfo{"VCVTSI2SS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2A), 1, X86InstInfo{"VCVTSI2SD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, X86InstInfo{"VMOVNTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2B), 1, X86InstInfo{"VMOVNTPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x2B), 1, X86InstInfo{"VMOVNTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2B), 1, X86InstInfo{"VMOVNTPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b10, 0x2C), 1, X86InstInfo{"VCVTTSS2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x2C), 1, X86InstInfo{"VCVTTSD2SI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -136,8 +136,8 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b00, 0x2F), 1, X86InstInfo{"VUCOMISS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x2F), 1, X86InstInfo{"VUCOMISD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, X86InstInfo{"VADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x58), 1, X86InstInfo{"VADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x58), 1, X86InstInfo{"VADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x58), 1, X86InstInfo{"VADDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x58), 1, X86InstInfo{"VADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x58), 1, X86InstInfo{"VADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -179,8 +179,8 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x6D), 1, X86InstInfo{"VPUNPCKHQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6E), 1, X86InstInfo{"VMOV*", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x6F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x6F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x6F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x6F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7C), 1, X86InstInfo{"VHADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7C), 1, X86InstInfo{"VHADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -188,11 +188,11 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0x7D), 1, X86InstInfo{"VHSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x7D), 1, X86InstInfo{"VHSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7E), 1, X86InstInfo{"VMOV*", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7E), 1, X86InstInfo{"VMOVQ", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7E), 1, X86InstInfo{"VMOV*", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7E), 1, X86InstInfo{"VMOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0x7F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x7F), 1, X86InstInfo{"VMOVDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x7F), 1, X86InstInfo{"VMOVDQU", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
{OPD(1, 0b01, 0xAE), 1, X86InstInfo{"", TYPE_VEX_GROUP_15, FLAGS_NONE, 0, nullptr}}, // VEX Group 15
|
||||
@@ -205,19 +205,19 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0xD1), 1, X86InstInfo{"VPSRLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD2), 1, X86InstInfo{"VPSRLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD3), 1, X86InstInfo{"VPSRLQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD4), 1, X86InstInfo{"VPADDQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD4), 1, X86InstInfo{"VPADDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD5), 1, X86InstInfo{"VPMULLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD6), 1, X86InstInfo{"VMOVQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD6), 1, X86InstInfo{"VMOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD7), 1, X86InstInfo{"VPMOVMSKB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_DST_GPR | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xD8), 1, X86InstInfo{"VPSUBUSB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xD9), 1, X86InstInfo{"VPSUBUSW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDA), 1, X86InstInfo{"VPMINUB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDB), 1, X86InstInfo{"VPAND", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDB), 1, X86InstInfo{"VPAND", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDC), 1, X86InstInfo{"VPADDUSB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDD), 1, X86InstInfo{"VPADDUSW", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDE), 1, X86InstInfo{"VPMAXUB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDF), 1, X86InstInfo{"VPANDN", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xDF), 1, X86InstInfo{"VPANDN", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE0), 1, X86InstInfo{"VPAVGB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE1), 1, X86InstInfo{"VPSRAW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -230,16 +230,16 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b10, 0xE6), 1, X86InstInfo{"VCVTDQ2PD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0xE6), 1, X86InstInfo{"VCVTPD2DQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE7), 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b01, 0xE8), 1, X86InstInfo{"VPSUBSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xE9), 1, X86InstInfo{"VPSUBSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEA), 1, X86InstInfo{"VPMINSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEB), 1, X86InstInfo{"VPOR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEA), 1, X86InstInfo{"VPMINSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEB), 1, X86InstInfo{"VPOR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEC), 1, X86InstInfo{"VPADDSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xED), 1, X86InstInfo{"VPADDSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEE), 1, X86InstInfo{"VPMAXSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEF), 1, X86InstInfo{"VPXOR", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEE), 1, X86InstInfo{"VPMAXSW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xEF), 1, X86InstInfo{"VPXOR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b11, 0xF0), 1, X86InstInfo{"VLDDQU", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -255,9 +255,9 @@ void InitializeVEXTables() {
|
||||
{OPD(1, 0b01, 0xF9), 1, X86InstInfo{"VPSUBW", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFA), 1, X86InstInfo{"VPSUBD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFB), 1, X86InstInfo{"VPSUBQ", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFC), 1, X86InstInfo{"VPADDB", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFD), 1, X86InstInfo{"VPADDW", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFE), 1, X86InstInfo{"VPADDD", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFC), 1, X86InstInfo{"VPADDB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFD), 1, X86InstInfo{"VPADDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0xFE), 1, X86InstInfo{"VPADDD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
// VEX Map 2
|
||||
{OPD(2, 0b01, 0x00), 1, X86InstInfo{"VPSHUFB", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -298,7 +298,7 @@ void InitializeVEXTables() {
|
||||
|
||||
{OPD(2, 0b01, 0x28), 1, X86InstInfo{"VPMULDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x29), 1, X86InstInfo{"VPCMPEQQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2A), 1, X86InstInfo{"VMOVNTDQA", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2A), 1, X86InstInfo{"VMOVNTDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2B), 1, X86InstInfo{"VPACKUSDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2C), 1, X86InstInfo{"VMASKMOVPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x2D), 1, X86InstInfo{"VMASKMOVPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
+4
-1
@@ -199,7 +199,7 @@ namespace FEXCore {
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
emit->_StoreContext(GPRSize, IR::GPRClass, emit->_Constant(Entrypoint), offsetof(Core::CPUState, gregs[X86State::REG_R11]));
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
@@ -431,6 +431,9 @@ namespace FEXCore {
|
||||
|
||||
FEX_DEFAULT_VISIBILITY
|
||||
void FinalizeHostTrampolineForGuestFunction(HostToGuestTrampolinePtr* TrampolineAddress, void* HostPacker) {
|
||||
|
||||
if (TrampolineAddress == nullptr) return;
|
||||
|
||||
auto& Trampoline = GetInstanceInfo(TrampolineAddress);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Trampoline.CallCallback == (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
|
||||
+14
-32
@@ -355,7 +355,9 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -370,7 +372,9 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -381,7 +385,9 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
"StoreContextIndexed SSA:$Value, GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
|
||||
@@ -393,7 +399,9 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
@@ -471,22 +479,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VLoadMemElement u8:#RegisterSize, u8:#ElementSize, FPR:$Value, GPR:$Addr, u8:$Index, u8:$Align{1}": {
|
||||
"Desc": ["Loads an element of size #ElementSize in to $Value from $Addr at $Index"
|
||||
],
|
||||
"OpClass": "Memory",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"VStoreMemElement u8:#RegisterSize, u8:#ElementSize, FPR:$Value, GPR:$Addr, u8:$Index, u8:$Align": {
|
||||
"Desc": ["Stores an element of size #ElementSize from $Value[$Index] to $Addr"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ElementSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"CacheLineClear GPR:$Addr": {
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified",
|
||||
"Only clears the data cachelines. Doesn't do any zeroing"
|
||||
@@ -1022,16 +1014,6 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSLI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$ByteShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VSRI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$ByteShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VShlI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1065,7 +1047,7 @@
|
||||
},
|
||||
"FPR = VSXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1077,7 +1059,7 @@
|
||||
},
|
||||
"FPR = VUXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Zero extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
|
||||
@@ -162,6 +162,12 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
else if (Arg == "FPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRClass};
|
||||
}
|
||||
else if (Arg == "GPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRFixedClass};
|
||||
}
|
||||
else if (Arg == "FPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRFixedClass};
|
||||
}
|
||||
else if (Arg == "GPRPair") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRPairClass};
|
||||
}
|
||||
|
||||
+3
-9
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
@@ -36,15 +37,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineCo
|
||||
|
||||
InsertPass(CreateSyscallOptimization());
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
else {
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
@@ -66,6 +58,8 @@ void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAV
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
bool Changed = false;
|
||||
for (auto const &Pass : Passes) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
|
||||
@@ -21,7 +21,6 @@ std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusiveP
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
|
||||
bool OptimizeSRA,
|
||||
bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
namespace Validation {
|
||||
|
||||
@@ -20,6 +20,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
@@ -1028,6 +1029,8 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
|
||||
bool ConstProp::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
@@ -22,6 +23,7 @@ private:
|
||||
};
|
||||
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
|
||||
|
||||
+132
-61
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
@@ -76,24 +77,6 @@ namespace {
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // SSE padding in non-AVX case
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
@@ -102,7 +85,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
sizeof(FEXCore::Core::CPUState::rip),
|
||||
},
|
||||
DefaultAccess[0],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -112,62 +95,134 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
|
||||
FEXCore::Core::CPUState::GPR_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[1],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es),
|
||||
sizeof(FEXCore::Core::CPUState::es),
|
||||
offsetof(FEXCore::Core::CPUState, es_idx),
|
||||
sizeof(FEXCore::Core::CPUState::es_idx),
|
||||
},
|
||||
DefaultAccess[2],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs),
|
||||
sizeof(FEXCore::Core::CPUState::cs),
|
||||
offsetof(FEXCore::Core::CPUState, cs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::cs_idx),
|
||||
},
|
||||
DefaultAccess[3],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss),
|
||||
sizeof(FEXCore::Core::CPUState::ss),
|
||||
offsetof(FEXCore::Core::CPUState, ss_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ss_idx),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds),
|
||||
sizeof(FEXCore::Core::CPUState::ds),
|
||||
offsetof(FEXCore::Core::CPUState, ds_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ds_idx),
|
||||
},
|
||||
DefaultAccess[5],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::gs_idx),
|
||||
},
|
||||
DefaultAccess[6],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
sizeof(FEXCore::Core::CPUState::fs),
|
||||
offsetof(FEXCore::Core::CPUState, fs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::fs_idx),
|
||||
},
|
||||
DefaultAccess[7],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad),
|
||||
sizeof(FEXCore::Core::CPUState::_pad),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es_cached),
|
||||
sizeof(FEXCore::Core::CPUState::es_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::cs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ss_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ds_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::gs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::fs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -178,7 +233,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -189,7 +244,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -199,7 +254,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
DefaultAccess[9],
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -210,7 +265,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
|
||||
FEXCore::Core::CPUState::FLAG_SIZE,
|
||||
},
|
||||
DefaultAccess[10],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -221,7 +276,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
|
||||
FEXCore::Core::CPUState::MM_REG_SIZE
|
||||
},
|
||||
DefaultAccess[11],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -233,7 +288,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::gdt[0]),
|
||||
},
|
||||
DefaultAccess[12],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -244,7 +299,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FCW),
|
||||
sizeof(FEXCore::Core::CPUState::FCW),
|
||||
},
|
||||
DefaultAccess[13],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -254,7 +309,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FTW),
|
||||
sizeof(FEXCore::Core::CPUState::FTW),
|
||||
},
|
||||
DefaultAccess[14],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -288,40 +343,55 @@ namespace {
|
||||
ContextClassification->at(Offset).StoreNode = nullptr;
|
||||
};
|
||||
size_t Offset = 0;
|
||||
SetAccess(Offset++, DefaultAccess[0]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[1]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[2]);
|
||||
SetAccess(Offset++, DefaultAccess[3]);
|
||||
SetAccess(Offset++, DefaultAccess[4]);
|
||||
SetAccess(Offset++, DefaultAccess[5]);
|
||||
SetAccess(Offset++, DefaultAccess[6]);
|
||||
SetAccess(Offset++, DefaultAccess[7]);
|
||||
// Segment indexes
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
// Segments
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad2
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[8]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
if (!SupportsAVX) {
|
||||
SetAccess(Offset++, DefaultAccess[9]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[10]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[11]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[12]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
SetAccess(Offset++, DefaultAccess[14]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
@@ -695,6 +765,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
|
||||
// XXX: We don't do cross-block optimizations yet
|
||||
//CalculateControlFlowInfo(IREmit);
|
||||
bool Changed = false;
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
@@ -154,6 +155,8 @@ struct Info {
|
||||
*
|
||||
*/
|
||||
bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
std::unordered_map<OrderedNode*, Info> InfoMap;
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
@@ -52,6 +53,8 @@ IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
}
|
||||
|
||||
bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRCompaction");
|
||||
|
||||
LocalBuilder.ReownOrClaimBuffer();
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
@@ -32,6 +33,8 @@ IRValidation::~IRValidation() {
|
||||
}
|
||||
|
||||
bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRValidation");
|
||||
|
||||
bool HadError = false;
|
||||
bool HadWarning = false;
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -53,6 +54,8 @@ bool LongDivideEliminationPass::IsSextOp(IREmitter *IREmit, OrderedNodeWrapper L
|
||||
}
|
||||
|
||||
bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::LDE");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
@@ -24,6 +25,8 @@ public:
|
||||
};
|
||||
|
||||
bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::PHIValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <deque>
|
||||
@@ -191,6 +191,8 @@ private:
|
||||
bool RAValidation::Run(IREmitter *IREmit) {
|
||||
if (!Manager->HasPass("RA")) return false;
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
BlockExitState.clear();
|
||||
// BlocksToVisit will already be empty
|
||||
|
||||
+4
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <array>
|
||||
@@ -32,6 +34,8 @@ public:
|
||||
*
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
std::array<OrderedNode*, 32> LastValidFlagStores{};
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -15,6 +15,8 @@ $end_info$
|
||||
#include <FEXCore/Utils/BucketList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -1527,6 +1529,7 @@ namespace {
|
||||
}
|
||||
|
||||
bool ConstrainedRAPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RA");
|
||||
bool Changed = false;
|
||||
|
||||
auto IR = IREmit->ViewIR();
|
||||
|
||||
-128
@@ -1,128 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Replaces Load/StoreContext with Load/StoreReg for SRA regs
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class StaticRegisterAllocationPass final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit StaticRegisterAllocationPass(bool SupportsAVX_) : SupportsAVX{SupportsAVX_} {}
|
||||
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
|
||||
private:
|
||||
bool SupportsAVX;
|
||||
|
||||
bool IsStaticAllocGpr(uint32_t Offset, RegisterClassType Class) const {
|
||||
const auto begin = offsetof(Core::CPUState, gregs[0]);
|
||||
const auto end = offsetof(Core::CPUState, gregs[16]);
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto reg = (Offset - begin) / Core::CPUState::GPR_REG_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(Class.Val == IR::GPRClass.Val, "unexpected Class {}", Class);
|
||||
|
||||
// 0..15 -> 16 in total
|
||||
return reg < Core::CPUState::NUM_GPRS;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool IsStaticAllocFpr(uint32_t Offset, RegisterClassType Class, bool AllowGpr) const {
|
||||
const auto [begin, end] = [this]() -> std::pair<ptrdiff_t, ptrdiff_t> {
|
||||
if (SupportsAVX) {
|
||||
return {
|
||||
offsetof(Core::CPUState, xmm.avx.data[0][0]),
|
||||
offsetof(Core::CPUState, xmm.avx.data[16][0]),
|
||||
};
|
||||
} else {
|
||||
return {
|
||||
offsetof(Core::CPUState, xmm.sse.data[0][0]),
|
||||
offsetof(Core::CPUState, xmm.sse.data[16][0]),
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto size = SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto reg = (Offset - begin) / size;
|
||||
LOGMAN_THROW_AA_FMT(Class.Val == IR::FPRClass.Val || (AllowGpr && Class.Val == IR::GPRClass.Val), "unexpected Class {}, AllowGpr {}", Class, AllowGpr);
|
||||
|
||||
// 0..15 -> 16 in total
|
||||
return reg < Core::CPUState::NUM_XMMS;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This pass replaces Load/Store Context with Load/Store Register for Statically Mapped registers. It also does some validation.
|
||||
*
|
||||
*/
|
||||
bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
|
||||
if (IsStaticAllocGpr(Op->Offset, Op->Class) || IsStaticAllocFpr(Op->Offset, Op->Class, true)) {
|
||||
auto GeneralClass = Op->Class;
|
||||
if (IsStaticAllocFpr(Op->Offset, GeneralClass, true) && GeneralClass == GPRClass) {
|
||||
GeneralClass = FPRClass;
|
||||
}
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
OrderedNode *sraReg = IREmit->_LoadRegister(false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
if (GeneralClass != Op->Class) {
|
||||
sraReg = IREmit->_VExtractToGPR(Op->Header.Size, Op->Header.Size, sraReg, 0);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, sraReg);
|
||||
}
|
||||
} if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
|
||||
if (IsStaticAllocGpr(Op->Offset, Op->Class) || IsStaticAllocFpr(Op->Offset, Op->Class, true)) {
|
||||
auto val = IREmit->UnwrapNode(Op->Value);
|
||||
|
||||
auto GeneralClass = Op->Class;
|
||||
if (IsStaticAllocFpr(Op->Offset, GeneralClass, true) && GeneralClass == GPRClass) {
|
||||
val = IREmit->_VCastFromGPR(Op->Header.Size, Op->Header.Size, val);
|
||||
GeneralClass = FPRClass;
|
||||
}
|
||||
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
IREmit->_StoreRegister(val, false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX) {
|
||||
return std::make_unique<StaticRegisterAllocationPass>(SupportsAVX);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -23,6 +24,8 @@ public:
|
||||
};
|
||||
|
||||
bool SyscallOptimization::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
@@ -36,6 +37,8 @@ public:
|
||||
};
|
||||
|
||||
bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ValueDominanceValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
+31
-7
@@ -145,8 +145,10 @@ namespace FEXCore::Allocator {
|
||||
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
|
||||
|
||||
std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
void * const StackLocation = alloca(0);
|
||||
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(StackLocation);
|
||||
std::vector<MemoryRegion> Regions;
|
||||
|
||||
|
||||
int MapsFD = open("/proc/self/maps", O_RDONLY);
|
||||
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
|
||||
|
||||
@@ -155,6 +157,8 @@ namespace FEXCore::Allocator {
|
||||
uintptr_t RegionBegin = 0;
|
||||
uintptr_t RegionEnd = 0;
|
||||
|
||||
uintptr_t PreviousMapEnd = 0;
|
||||
|
||||
char Buffer[2048];
|
||||
const char *Cursor;
|
||||
ssize_t Remaining = 0;
|
||||
@@ -162,7 +166,7 @@ namespace FEXCore::Allocator {
|
||||
for(;;) {
|
||||
|
||||
if (Remaining == 0) {
|
||||
do {
|
||||
do {
|
||||
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
|
||||
} while ( Remaining == -1 && errno == EAGAIN);
|
||||
|
||||
@@ -172,8 +176,8 @@ namespace FEXCore::Allocator {
|
||||
if (Remaining == 0 && State == ParseBegin) {
|
||||
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = End;
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = End;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
@@ -209,9 +213,12 @@ namespace FEXCore::Allocator {
|
||||
if (c == '-') {
|
||||
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
// Store the location we are going to map.
|
||||
PreviousMapEnd = MapEnd;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
if (MapEnd > MapBegin) {
|
||||
@@ -225,6 +232,7 @@ namespace FEXCore::Allocator {
|
||||
|
||||
Regions.push_back({(void*)MapBegin, MapSize});
|
||||
}
|
||||
|
||||
RegionBegin = 0;
|
||||
RegionEnd = 0;
|
||||
State = ParseEnd;
|
||||
@@ -240,6 +248,22 @@ namespace FEXCore::Allocator {
|
||||
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
State = ScanEnd;
|
||||
|
||||
// If the previous map's ending and the region we just parsed overlap the stack then we need to save the stack mapping.
|
||||
// Otherwise we will have severely limited stack size which crashes quickly.
|
||||
if (PreviousMapEnd <= StackLocation_u64 && RegionEnd > StackLocation_u64) {
|
||||
auto BelowStackRegion = Regions.back();
|
||||
LOGMAN_THROW_AA_FMT(reinterpret_cast<uint64_t>(BelowStackRegion.Ptr) + BelowStackRegion.Size == PreviousMapEnd,
|
||||
"This needs to match");
|
||||
|
||||
// Allocate the region under the stack as READ | WRITE so the stack can still grow
|
||||
auto Alloc = mmap(BelowStackRegion.Ptr, BelowStackRegion.Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", BelowStackRegion.Ptr, BelowStackRegion.Size);
|
||||
LogMan::Throw::AFmt(Alloc == BelowStackRegion.Ptr, "mmap({},{:x}) returned {} instead of {:x}", Alloc, BelowStackRegion.Ptr);
|
||||
|
||||
Regions.pop_back();
|
||||
}
|
||||
continue;
|
||||
} else {
|
||||
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
|
||||
|
||||
+83
-102
@@ -69,11 +69,14 @@ namespace Alloc::OSAllocator {
|
||||
struct LiveVMARegion {
|
||||
ReservedVMARegion *SlabInfo;
|
||||
uint64_t FreeSpace{};
|
||||
uint64_t NumManagedPages{};
|
||||
uint32_t LastPageAllocation{};
|
||||
bool HadMunmap{};
|
||||
|
||||
// Align UsedPages so it pads to the next page.
|
||||
// Necessary to take advantage of madvise zero page pooling.
|
||||
alignas(4096) FEXCore::FlexBitSet<uint64_t> UsedPages;
|
||||
using FlexBitElementType = uint64_t;
|
||||
alignas(4096) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
|
||||
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
|
||||
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
|
||||
@@ -85,8 +88,8 @@ namespace Alloc::OSAllocator {
|
||||
// 0x100'0000 Pages
|
||||
// 1 bit per page for tracking means 0x20'0000 (Pages / 8) bytes of flex space
|
||||
// Which is 2MB of tracking
|
||||
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(uint64_t);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<uint64_t>::Size(NumElements);
|
||||
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::Size(NumElements);
|
||||
}
|
||||
|
||||
static void InitializeVMARegionUsed(LiveVMARegion *Region, size_t AdditionalSize) {
|
||||
@@ -95,19 +98,21 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
|
||||
|
||||
size_t NumPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
|
||||
size_t NumManagedPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
|
||||
size_t ManagedSize = NumManagedPages << FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
// Use madvise to set the full tracking region to zero.
|
||||
// This ensures unused pages are zero, while not having the backing pages consuming memory.
|
||||
::madvise(Region->UsedPages.Memory + (NumPages * 4096), (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - (NumPages * 4096), MADV_DONTNEED);
|
||||
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - ManagedSize, MADV_DONTNEED);
|
||||
|
||||
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
|
||||
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
|
||||
::madvise(Region->UsedPages.Memory, NumPages * 4096, MADV_WILLNEED);
|
||||
::madvise(Region->UsedPages.Memory, ManagedSize, MADV_WILLNEED);
|
||||
|
||||
// Set our reserved pages
|
||||
Region->UsedPages.MemSet(NumPages);
|
||||
Region->LastPageAllocation = NumPages;
|
||||
Region->UsedPages.MemSet(NumManagedPages);
|
||||
Region->LastPageAllocation = NumManagedPages;
|
||||
Region->NumManagedPages = NumManagedPages;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -129,6 +134,7 @@ namespace Alloc::OSAllocator {
|
||||
ReservedVMARegion *ReservedRegion = *ReservedIterator;
|
||||
|
||||
ReservedRegions->erase(ReservedIterator);
|
||||
|
||||
// mprotect the new region we've allocated
|
||||
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FHU::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
|
||||
@@ -152,6 +158,9 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
// 32-bit old kernel workarounds
|
||||
std::vector<FEXCore::Allocator::MemoryRegion> Steal32BitIfOldKernel();
|
||||
|
||||
void AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges);
|
||||
LiveVMARegion *FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd);
|
||||
};
|
||||
|
||||
void OSAllocator_64Bit::DetermineVASize() {
|
||||
@@ -167,6 +176,42 @@ void OSAllocator_64Bit::DetermineVASize() {
|
||||
UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::LiveVMARegion *OSAllocator_64Bit::FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd) {
|
||||
LiveVMARegion *LiveRegion{};
|
||||
|
||||
// Check active slabs to see if we can fit this
|
||||
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin &&
|
||||
Addr < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Couldn't find an active region that fit
|
||||
// Check reserved regions
|
||||
if (!LiveRegion) {
|
||||
// Didn't have a slab that fit this range
|
||||
// Check our reserved regions to see if we have one that fits
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
ReservedVMARegion *ReservedRegion = *it;
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base &&
|
||||
AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return LiveRegion;
|
||||
}
|
||||
|
||||
void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
if (addr != 0 &&
|
||||
addr < reinterpret_cast<void*>(LOWER_BOUND)) {
|
||||
@@ -205,41 +250,13 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
LiveVMARegion *LiveRegion{};
|
||||
|
||||
if (Fixed || Addr != 0) {
|
||||
// Check active slabs to see if we can fit this
|
||||
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin &&
|
||||
Addr < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Couldn't find an active region that fit
|
||||
// Check reserved regions
|
||||
if (!LiveRegion) {
|
||||
// Didn't have a slab that fit this range
|
||||
// Check our reserved regions to see if we have one that fits
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
ReservedVMARegion *ReservedRegion = *it;
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base &&
|
||||
AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
LiveRegion = FindLiveRegionForAddress(Addr, AddrEnd);
|
||||
}
|
||||
|
||||
again:
|
||||
|
||||
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion *Region, uint64_t length, int prot, int flags, int fd, off_t offset, uint64_t StartingPosition = 0) -> std::pair<LiveVMARegion*, void*> {
|
||||
uint64_t AllocatedPage{};
|
||||
uint64_t AllocatedPage{~0ULL};
|
||||
uint64_t NumberOfPages = length >> FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
if (Region->FreeSpace >= length) {
|
||||
@@ -249,72 +266,29 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
: Region->LastPageAllocation;
|
||||
size_t RegionNumberOfPages = Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
// Backward scan
|
||||
// We need to do a backward scan first to fill any holes
|
||||
// Otherwise we will very quickly run out of VMA regions (65k maximum)
|
||||
for (size_t CurrentPage = LastAllocation;
|
||||
CurrentPage >= NumberOfPages;) {
|
||||
size_t Remaining = NumberOfPages;
|
||||
assert(Remaining <= CurrentPage);
|
||||
|
||||
while (Remaining) {
|
||||
if (Region->UsedPages[CurrentPage - Remaining]) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
if (Region->HadMunmap) {
|
||||
// Backward scan
|
||||
// We need to do a backward scan first to fill any holes
|
||||
// Otherwise we will very quickly run out of VMA regions (65k maximum)
|
||||
auto SearchResult = Region->UsedPages.BackwardScanForRange<true>(LastAllocation, NumberOfPages, Region->NumManagedPages);
|
||||
|
||||
if (Remaining) {
|
||||
// Didn't find a slab range
|
||||
CurrentPage -= Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
CurrentPage -= NumberOfPages;
|
||||
AllocatedPage = SearchResult.FoundElement;
|
||||
|
||||
// Keep scanning backwards to not introduce ANOTHER gap
|
||||
while (CurrentPage >= 1) {
|
||||
if (Region->UsedPages[CurrentPage - 1]) {
|
||||
// Found a used page, we can leave now
|
||||
break;
|
||||
}
|
||||
--CurrentPage;
|
||||
}
|
||||
AllocatedPage = CurrentPage;
|
||||
break;
|
||||
// If we didn't even have a one page free in the backward search, then unclaim HadMunmap.
|
||||
// Switching over to default forward search.
|
||||
if (SearchResult.FoundElement == ~0ULL && !SearchResult.FoundHole) {
|
||||
Region->HadMunmap = false;
|
||||
}
|
||||
}
|
||||
|
||||
// Foward Scan
|
||||
if (AllocatedPage == 0) {
|
||||
for (size_t CurrentPage = LastAllocation;
|
||||
CurrentPage < (RegionNumberOfPages - NumberOfPages);) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = NumberOfPages;
|
||||
|
||||
assert((CurrentPage + Remaining - 1) < RegionNumberOfPages);
|
||||
while (Remaining) {
|
||||
if (Region->UsedPages[CurrentPage + Remaining - 1]) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// Didn't find a slab range
|
||||
CurrentPage += Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
AllocatedPage = CurrentPage;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (AllocatedPage == ~0ULL) {
|
||||
auto SearchResult = Region->UsedPages.ForwardScanForRange<true>(LastAllocation, NumberOfPages, RegionNumberOfPages);
|
||||
AllocatedPage = SearchResult.FoundElement;
|
||||
}
|
||||
|
||||
if (AllocatedPage) {
|
||||
if (AllocatedPage != ~0ULL) {
|
||||
AllocatedOffset = Region->SlabInfo->Base + AllocatedPage * FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// We need to setup protections for this
|
||||
@@ -497,6 +471,8 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
// This will let us more quickly fill holes
|
||||
(*it)->LastPageAllocation = std::min((*it)->LastPageAllocation, SlabPageBegin);
|
||||
|
||||
(*it)->HadMunmap = true;
|
||||
|
||||
// XXX: Move region back to reserved list
|
||||
return 0;
|
||||
}
|
||||
@@ -537,12 +513,7 @@ std::vector<FEXCore::Allocator::MemoryRegion> OSAllocator_64Bit::Steal32BitIfOld
|
||||
return FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND_32, UPPER_BOUND_32);
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
DetermineVASize();
|
||||
auto LowMem = Steal32BitIfOldKernel();
|
||||
|
||||
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
|
||||
|
||||
void OSAllocator_64Bit::AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges) {
|
||||
for (auto [Ptr, AllocationSize]: Ranges) {
|
||||
if (!ObjectAlloc) {
|
||||
auto MaxSize = std::min(size_t(64) * 1024 * 1024, AllocationSize);
|
||||
@@ -564,12 +535,22 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
ReservedVMARegion *Region = ObjectAlloc->new_construct<ReservedVMARegion>();
|
||||
Region->Base = reinterpret_cast<uint64_t>(Ptr);
|
||||
Region->RegionSize = AllocationSize;
|
||||
ReservedRegions->emplace_back(Region);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
DetermineVASize();
|
||||
auto LowMem = Steal32BitIfOldKernel();
|
||||
|
||||
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
|
||||
|
||||
AllocateMemoryRegions(Ranges);
|
||||
|
||||
FEXCore::Allocator::ReclaimMemoryRegion(LowMem);
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -38,6 +39,109 @@ struct FlexBitSet final {
|
||||
memset(Memory, 0xFF, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
}
|
||||
|
||||
// Range scanning results
|
||||
struct BitsetScanResults {
|
||||
// Which element was found. ~0ULL if not found.
|
||||
size_t FoundElement;
|
||||
// During the scan, found a hole in the allocations that didn't fit.
|
||||
bool FoundHole;
|
||||
};
|
||||
|
||||
// TODO: Make {Forward,Backward}ScanForRange faster
|
||||
// Currently these functions test a single bit at a time, which is fairly costly.
|
||||
// The compiler emits a full element load per iteration, wasting a bunch of time on loads.
|
||||
// If we change these functions to have a pre-amble and post-amble to align the primary loop to the element size then this can go significantly
|
||||
// faster.
|
||||
//
|
||||
// Once the element scanning is aligned to the element size, we can then use native count leading zero(CLZ) and count trailing zero(CTZ)
|
||||
// instructions on a full element to scan uint64_t elements per loop iteration.
|
||||
|
||||
// Implementation details:
|
||||
// Template argument WantUnset
|
||||
// Used to determine if the desired range is for set or unset ranges.
|
||||
// Typically `WantUnset` should be true. Used for finding a unset range inside of a range will set elements.
|
||||
//
|
||||
// @param BeginningElement - The first element in the set to start scanning from.
|
||||
// @param ElementCount - How many elements to find a range for fitting.
|
||||
// @param MinimumElement - Minimum element in the set to search to
|
||||
//
|
||||
// @return The scan results
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults BackwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t MinimumElement) {
|
||||
bool FoundHole {};
|
||||
for (size_t CurrentPage = BeginningElement;
|
||||
CurrentPage >= (MinimumElement + ElementCount);) {
|
||||
size_t Remaining = ElementCount;
|
||||
LOGMAN_THROW_AA_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentPage - Remaining) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// If we found at least one Element hole then track that
|
||||
if (Remaining != ElementCount) {
|
||||
FoundHole = true;
|
||||
}
|
||||
|
||||
// Didn't find a slab range
|
||||
CurrentPage -= Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults{CurrentPage - ElementCount, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
return BitsetScanResults {~0ULL, FoundHole};
|
||||
}
|
||||
|
||||
// @param BeginningElement - The first element in the set to start scanning from.
|
||||
// @param ElementCount - How many elements to find a range for fitting.
|
||||
// @param ElementsInSet - How many elements are in the full set.
|
||||
//
|
||||
// @return The scan results
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults ForwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t ElementsInSet) {
|
||||
bool FoundHole {};
|
||||
|
||||
for (size_t CurrentElement = BeginningElement;
|
||||
CurrentElement < (ElementsInSet - ElementCount);) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = ElementCount;
|
||||
|
||||
LOGMAN_THROW_AA_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// If we found at least one Element hole then track that
|
||||
if (Remaining != ElementCount) {
|
||||
FoundHole = true;
|
||||
}
|
||||
|
||||
// Didn't find a slab range
|
||||
CurrentElement += Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults {CurrentElement, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
return BitsetScanResults {~0ULL, FoundHole};
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
bool operator[](size_t Element) const {
|
||||
|
||||
+116
@@ -0,0 +1,116 @@
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <linux/magic.h>
|
||||
#include <string>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/vfs.h>
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#define BACKEND_OFF 0
|
||||
#define BACKEND_GPUVIS 1
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
namespace FEXCore::Profiler {
|
||||
ProfilerBlock::ProfilerBlock(std::string_view const Format)
|
||||
: DurationBegin {GetTime()}
|
||||
, Format {Format} {
|
||||
}
|
||||
|
||||
ProfilerBlock::~ProfilerBlock() {
|
||||
auto Duration = GetTime() - DurationBegin;
|
||||
TraceObject(Format, Duration);
|
||||
}
|
||||
}
|
||||
|
||||
namespace GPUVis {
|
||||
// ftrace FD for writing trace data.
|
||||
// Needs to be a raw FD since we hold this open for the entire application execution.
|
||||
static int TraceFD {-1};
|
||||
|
||||
// Need to search the paths to find the real trace path
|
||||
static std::array<char const*, 2> TraceFSDirectories {
|
||||
"/sys/kernel/tracing",
|
||||
"/sys/kernel/debug/tracing",
|
||||
};
|
||||
|
||||
static bool IsTraceFS(char const* Path) {
|
||||
struct statfs stat;
|
||||
if (statfs(Path, &stat)) {
|
||||
return false;
|
||||
}
|
||||
return stat.f_type == TRACEFS_MAGIC;
|
||||
}
|
||||
|
||||
void Init() {
|
||||
for (auto Path : TraceFSDirectories) {
|
||||
if (IsTraceFS(Path)) {
|
||||
std::string FilePath = fmt::format("{}/trace_marker", Path);
|
||||
TraceFD = open(FilePath.c_str(), O_WRONLY | O_CLOEXEC);
|
||||
if (TraceFD != -1) {
|
||||
// Opened TraceFD, early exit
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
if (TraceFD != -1) {
|
||||
close(TraceFD);
|
||||
TraceFD = -1;
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
if (TraceFD != -1) {
|
||||
// Print the duration as something that began negative duration ago
|
||||
std::string Event = fmt::format("{} (lduration=-{})\n", Format, Duration);
|
||||
write(TraceFD, Event.c_str(), Event.size());
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
if (TraceFD != -1) {
|
||||
std::string Event = fmt::format("{}\n", Format);
|
||||
write(TraceFD, Format.data(), Format.size());
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
#error Unknown profiler backend
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
void Init() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::Init();
|
||||
#endif
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::Shutdown();
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format, Duration);
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format);
|
||||
#endif
|
||||
|
||||
}
|
||||
#endif
|
||||
}
|
||||
+3
-1
@@ -74,7 +74,9 @@ namespace Handler {
|
||||
LAYER_GLOBAL_MAIN, ///< /usr/share/fex-emu/Config.json by default
|
||||
LAYER_MAIN,
|
||||
LAYER_ARGUMENTS,
|
||||
LAYER_GLOBAL_STEAM_APP,
|
||||
LAYER_GLOBAL_APP,
|
||||
LAYER_LOCAL_STEAM_APP,
|
||||
LAYER_LOCAL_APP,
|
||||
LAYER_ENVIRONMENT,
|
||||
LAYER_TOP,
|
||||
@@ -272,7 +274,7 @@ namespace Type {
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
|
||||
+18
-7
@@ -28,9 +28,16 @@ namespace FEXCore::Core {
|
||||
|
||||
uint64_t rip; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16];
|
||||
uint16_t es, cs, ss, ds;
|
||||
uint64_t gs;
|
||||
uint64_t fs;
|
||||
// Raw segment register indexes
|
||||
uint16_t es_idx, cs_idx, ss_idx, ds_idx;
|
||||
uint16_t gs_idx, fs_idx;
|
||||
uint16_t _pad[2];
|
||||
|
||||
// Segment registers holding base addresses
|
||||
uint32_t es_cached, cs_cached, ss_cached, ds_cached;
|
||||
uint64_t gs_cached;
|
||||
uint64_t fs_cached;
|
||||
uint64_t _pad2[1];
|
||||
XMMRegs xmm;
|
||||
uint8_t flags[48];
|
||||
uint64_t mm[8][2];
|
||||
@@ -211,12 +218,13 @@ namespace FEXCore::Core {
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
|
||||
struct SynchronousFaultDataStruct {
|
||||
struct alignas(8) SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint8_t Signal;
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
uint8_t TrapNo;
|
||||
uint8_t si_code;
|
||||
uint16_t err_code;
|
||||
uint32_t _pad : 16;
|
||||
} SynchronousFaultData;
|
||||
|
||||
InternalThreadState* Thread;
|
||||
@@ -230,6 +238,9 @@ namespace FEXCore::Core {
|
||||
static_assert(offsetof(CpuStateFrame, Pointers) + sizeof(CpuStateFrame::Pointers) <= 32760, "JITPointers maximum pointer needs to be less than architecture maximum 32768");
|
||||
|
||||
static_assert(std::is_standard_layout<CpuStateFrame>::value, "This needs to be standard layout");
|
||||
static_assert(sizeof(CpuStateFrame::SynchronousFaultData) == 8, "This needs to be 8 bytes");
|
||||
static_assert(std::alignment_of_v<CpuStateFrame::SynchronousFaultDataStruct> == 8, "This needs to be 8 bytes");
|
||||
static_assert(offsetof(CpuStateFrame, SynchronousFaultData) % 8 == 0, "This needs to be aligned");
|
||||
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
|
||||
|
||||
@@ -27,6 +27,7 @@ class HostFeatures final {
|
||||
bool SupportsSHA{};
|
||||
bool SupportsBMI1{};
|
||||
bool SupportsBMI2{};
|
||||
bool SupportsPMULL_128Bit{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
|
||||
@@ -8,8 +8,8 @@
|
||||
#include <FEXCore/Utils/InterruptableConditionVariable.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <map>
|
||||
#include <unordered_map>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -101,7 +101,7 @@ namespace FEXCore::Core {
|
||||
std::unique_ptr<FEXCore::CPU::CPUBackend> CPUBackend;
|
||||
std::unique_ptr<FEXCore::LookupCache> LookupCache;
|
||||
|
||||
std::unordered_map<uint64_t, LocalIREntry> DebugStore;
|
||||
tsl::robin_map<uint64_t, LocalIREntry> DebugStore;
|
||||
|
||||
std::unique_ptr<FEXCore::Frontend::Decoder> FrontendDecoder;
|
||||
std::unique_ptr<FEXCore::IR::PassManager> PassManager;
|
||||
@@ -115,7 +115,7 @@ namespace FEXCore::Core {
|
||||
|
||||
std::shared_mutex ObjectCacheRefCounter{};
|
||||
bool DestroyedByParent{false}; // Should the parent destroy this thread, or it destory itself
|
||||
|
||||
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
|
||||
|
||||
};
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#pragma once
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
|
||||
@@ -280,6 +280,11 @@ public:
|
||||
return Wrapper.GetNode(GetListData());
|
||||
}
|
||||
|
||||
///< Gets an OrderedNode from the IRListView as an OrderedNodeWrapper.
|
||||
[[nodiscard]] OrderedNodeWrapper WrapNode(OrderedNode *Node) const {
|
||||
return Node->Wrapped(GetListData());
|
||||
}
|
||||
|
||||
private:
|
||||
struct BlockRange {
|
||||
using iterator = NodeIterator;
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
#include <time.h>
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
|
||||
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
// clock_gettime will do a VDSO call with the least amount of overhead
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return ts.tv_sec * 1'000'000'000ULL + ts.tv_nsec;
|
||||
}
|
||||
|
||||
// A class that follows scoping rules to generate a profile duration block
|
||||
class ProfilerBlock final {
|
||||
public:
|
||||
ProfilerBlock(std::string_view const Format);
|
||||
|
||||
~ProfilerBlock();
|
||||
|
||||
private:
|
||||
uint64_t DurationBegin;
|
||||
std::string_view const Format;
|
||||
};
|
||||
|
||||
#define UniqueScopeName2(name, line) name ## line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) \
|
||||
FEXCore::Profiler::ProfilerBlock UniqueScopeName(ScopedBlock_, __LINE__) (name)
|
||||
|
||||
#else
|
||||
[[maybe_unused]] static void Init() {}
|
||||
[[maybe_unused]] static void Shutdown() {}
|
||||
[[maybe_unused]] static void TraceObject(std::string_view const Format) {}
|
||||
[[maybe_unused]] static void TraceObject(std::string_view const, uint64_t) {}
|
||||
|
||||
#define FEXCORE_PROFILE_INSTANT(...) do {} while(0)
|
||||
#define FEXCORE_PROFILE_SCOPED(...) do {} while(0)
|
||||
#endif
|
||||
}
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 2b55157592...98f440ce68.
Vendored
+1
-1
Submodule External/vixl updated: bbc7bdc609...af65c2974e.
@@ -148,16 +148,16 @@ def HandleFunctionDeclCursor(Arch, Cursor):
|
||||
elif (Child.kind == CursorKind.PARM_DECL):
|
||||
# This gives us a parameter type
|
||||
Function.Params.append(Child.type.spelling)
|
||||
elif (Child.kind == CursorKind.UNEXPOSED_ATTR):
|
||||
# Whatever you are we don't care about you
|
||||
return Arch
|
||||
elif (Child.kind == CursorKind.ASM_LABEL_ATTR):
|
||||
# Whatever you are we don't care about you
|
||||
return Arch
|
||||
elif (Child.kind == CursorKind.WARN_UNUSED_RESULT_ATTR):
|
||||
# Whatever you are we don't care about you
|
||||
return Arch
|
||||
elif (Child.kind == CursorKind.VISIBILITY_ATTR):
|
||||
elif (Child.kind == CursorKind.VISIBILITY_ATTR or
|
||||
Child.kind == CursorKind.UNEXPOSED_ATTR or
|
||||
Child.kind == CursorKind.CONST_ATTR or
|
||||
Child.kind == CursorKind.PURE_ATTR):
|
||||
pass
|
||||
else:
|
||||
logging.critical ("\tUnhandled FunctionDeclCursor {0}-{1}-{2}".format(Child.kind, Child.type.spelling, Child.spelling))
|
||||
|
||||
@@ -82,9 +82,8 @@ def IsSupportedDistro():
|
||||
if Distro[0] == "ubuntu":
|
||||
# We only support what is available in ppa:fex-emu/fex
|
||||
return Distro[1] == "20.04" or \
|
||||
Distro[1] == "21.04" or \
|
||||
Distro[1] == "21.10" or \
|
||||
Distro[1] == "22.04"
|
||||
Distro[1] == "22.04" or \
|
||||
Distro[1] == "22.10"
|
||||
|
||||
return False
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@ import subprocess
|
||||
import os.path
|
||||
from os import path
|
||||
|
||||
# Args: <Known Failures file> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
|
||||
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
|
||||
|
||||
if (len(sys.argv) < 7):
|
||||
sys.exit()
|
||||
@@ -12,19 +12,25 @@ if (len(sys.argv) < 7):
|
||||
known_failures = {}
|
||||
disabled_tests = {}
|
||||
known_failures_file = sys.argv[1]
|
||||
disabled_tests_file = sys.argv[2]
|
||||
disabled_tests_type_file = sys.argv[3]
|
||||
disabled_tests_runner_file = sys.argv[4]
|
||||
known_failures_type_file = sys.argv[2]
|
||||
disabled_tests_file = sys.argv[3]
|
||||
disabled_tests_type_file = sys.argv[4]
|
||||
disabled_tests_runner_file = sys.argv[5]
|
||||
|
||||
current_test = sys.argv[5]
|
||||
runner = sys.argv[6]
|
||||
args_start_index = 7
|
||||
current_test = sys.argv[6]
|
||||
runner = sys.argv[7]
|
||||
args_start_index = 8
|
||||
|
||||
# Open the known failures file and add it to a dictionary
|
||||
with open(known_failures_file) as kff:
|
||||
for line in kff:
|
||||
known_failures[line.strip()] = 1
|
||||
|
||||
if path.exists(known_failures_type_file):
|
||||
with open(known_failures_type_file) as dtf:
|
||||
for line in dtf:
|
||||
known_failures[line.strip()] = 1
|
||||
|
||||
with open(disabled_tests_file) as dtf:
|
||||
for line in dtf:
|
||||
disabled_tests[line.strip()] = 1
|
||||
|
||||
@@ -109,8 +109,18 @@ namespace FEX::Config {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, true));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, false));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
|
||||
auto SteamID = getenv("SteamAppId");
|
||||
if (SteamID) {
|
||||
// If a SteamID exists then let's search for Steam application configs as well.
|
||||
// We want to key off both the SteamAppId number /and/ the executable since we may not want to thunk all binaries.
|
||||
auto SteamAppName = fmt::format("Steam_{}_{}", SteamID, ProgramName.string());
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
FEXCore::Config::AddLayer(FEXCore::Config::CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
}
|
||||
|
||||
return std::make_pair(Program, ProgramName);
|
||||
}
|
||||
return {};
|
||||
|
||||
@@ -100,33 +100,78 @@ namespace FEXServerClient {
|
||||
return GetServerLockFolder() + "RootFS.lock";
|
||||
}
|
||||
|
||||
std::string GetServerSocketFile() {
|
||||
FEX_CONFIG_OPT(ServerSocketPath, SERVERSOCKETPATH);
|
||||
if (ServerSocketPath().empty()) {
|
||||
return fmt::format("{}/{}.FEXServer.socket", std::filesystem::temp_directory_path().string(), ::geteuid());
|
||||
std::string GetServerMountFolder() {
|
||||
// We need a FEXServer mount directory that has some tricky requirements.
|
||||
// - We don't want to use `/tmp/` if possible.
|
||||
// - systemd services use `PrivateTmp` feature to gives services their own tmp.
|
||||
// - We will use this as a fallback path /only/.
|
||||
// - Can't be `[$XDG_DATA_HOME,$HOME]/.fex-emu/`
|
||||
// - Might be mounted with a filesystem (sshfs) which can't handle mount points inside it.
|
||||
//
|
||||
// Directories it can be in:
|
||||
// - $XDG_RUNTIME_DIR if set
|
||||
// - Is typically `/run/user/<UID>/`
|
||||
// - systemd `PrivateTmp` feature doesn't touch this.
|
||||
// - If this path doesn't exist then fallback to `/tmp/` as a last resort.
|
||||
// - pressure-vessel explicitly creates an internal XDG_RUNTIME_DIR inside its chroot.
|
||||
// - This is okay since pressure-vessel rbinds the FEX rootfs from the host to `/run/pressure-vessel/interpreter-root`.
|
||||
std::string Folder{};
|
||||
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
|
||||
if (XDGRuntimeEnv) {
|
||||
// If the XDG runtime directory works then use that.
|
||||
Folder = XDGRuntimeEnv;
|
||||
}
|
||||
else {
|
||||
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
|
||||
// Might not be ideal but we don't have much of a choice.
|
||||
Folder = std::filesystem::temp_directory_path().string();
|
||||
}
|
||||
|
||||
return ServerSocketPath;
|
||||
if (FEXCore::Config::FindContainer() == "pressure-vessel") {
|
||||
// In pressure-vessel the mount point changes location.
|
||||
// This is due to pressure-vesssel being a chroot environment.
|
||||
// It by default maps the host-filesystem to `/run/host/` so we need to redirect.
|
||||
// After pressure-vessel is fully set up it will set the `FEX_ROOTFS` environment variable,
|
||||
// which the FEXInterpreter will pick up on.
|
||||
Folder = "/run/host/" + Folder;
|
||||
}
|
||||
|
||||
return Folder;
|
||||
}
|
||||
|
||||
std::string GetServerSocketName() {
|
||||
return fmt::format("{}.FEXServer.Socket", ::geteuid());
|
||||
}
|
||||
|
||||
int GetServerFD() {
|
||||
return ServerFD;
|
||||
}
|
||||
|
||||
int ConnectToServer() {
|
||||
auto ServerSocketFile = GetServerSocketFile();
|
||||
int ConnectToServer(ConnectionOption ConnectionOption) {
|
||||
auto ServerSocketName = GetServerSocketName();
|
||||
|
||||
// Create the initial unix socket
|
||||
int SocketFD = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
if (SocketFD == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
return -1;
|
||||
}
|
||||
|
||||
// AF_UNIX has a special feature for named socket paths.
|
||||
// If the name of the socket begins with `\0` then it is an "abstract" socket address.
|
||||
// The entirety of the name is used as a path to a socket that doesn't have any filesystem backing.
|
||||
struct sockaddr_un addr{};
|
||||
addr.sun_family = AF_UNIX;
|
||||
strncpy(addr.sun_path, ServerSocketFile.data(), std::min(ServerSocketFile.size(), sizeof(addr.sun_path)));
|
||||
size_t SizeOfSocketString = std::min(ServerSocketName.size() + 1, sizeof(addr.sun_path) - 1);
|
||||
addr.sun_path[0] = 0; // Abstract AF_UNIX sockets start with \0
|
||||
strncpy(addr.sun_path + 1, ServerSocketName.data(), SizeOfSocketString);
|
||||
// Include final null character.
|
||||
size_t SizeOfAddr = sizeof(addr.sun_family) + SizeOfSocketString;
|
||||
|
||||
if (connect(SocketFD, reinterpret_cast<struct sockaddr*>(&addr), sizeof(addr)) == -1) {
|
||||
if (connect(SocketFD, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr) == -1) {
|
||||
if (ConnectionOption == ConnectionOption::Default || errno != ECONNREFUSED) {
|
||||
LogMan::Msg::EFmt("Couldn't connect to FEXServer socket {} {} {}", ServerSocketName, errno, strerror(errno));
|
||||
}
|
||||
close(SocketFD);
|
||||
return -1;
|
||||
}
|
||||
@@ -153,7 +198,7 @@ namespace FEXServerClient {
|
||||
}
|
||||
|
||||
int ConnectToAndStartServer(char *InterpreterPath) {
|
||||
int ServerFD = ConnectToServer();
|
||||
int ServerFD = ConnectToServer(ConnectionOption::NoPrintConnectionError);
|
||||
if (ServerFD == -1) {
|
||||
// Couldn't connect to the server. Start one
|
||||
|
||||
@@ -210,7 +255,7 @@ namespace FEXServerClient {
|
||||
while (poll(&PollFD, 1, -1) == -1 && errno == EINTR);
|
||||
|
||||
for (size_t i = 0; i < 5; ++i) {
|
||||
ServerFD = ConnectToServer();
|
||||
ServerFD = ConnectToServer(ConnectionOption::Default);
|
||||
|
||||
if (ServerFD != -1) {
|
||||
break;
|
||||
@@ -218,6 +263,11 @@ namespace FEXServerClient {
|
||||
|
||||
std::this_thread::sleep_for(std::chrono::seconds(1));
|
||||
}
|
||||
|
||||
if (ServerFD == -1) {
|
||||
// Still couldn't connect to the socket.
|
||||
LogMan::Msg::EFmt("Couldn't connect to FEXServer socket {} after launching the process", GetServerSocketName());
|
||||
}
|
||||
}
|
||||
}
|
||||
return ServerFD;
|
||||
|
||||
@@ -50,7 +50,8 @@ namespace FEXServerClient {
|
||||
std::string GetServerLockFolder();
|
||||
std::string GetServerLockFile();
|
||||
std::string GetServerRootFSLockFile();
|
||||
std::string GetServerSocketFile();
|
||||
std::string GetServerMountFolder();
|
||||
std::string GetServerSocketName();
|
||||
int GetServerFD();
|
||||
|
||||
bool SetupClient(char *InterpreterPath);
|
||||
@@ -62,12 +63,16 @@ namespace FEXServerClient {
|
||||
*/
|
||||
int ConnectToAndStartServer(char *InterpreterPath);
|
||||
|
||||
enum class ConnectionOption {
|
||||
Default,
|
||||
NoPrintConnectionError,
|
||||
};
|
||||
/**
|
||||
* @brief Connect to a FEXServer instance if it exists
|
||||
*
|
||||
* @return socket FD for communicating with server
|
||||
*/
|
||||
int ConnectToServer();
|
||||
int ConnectToServer(ConnectionOption ConnectionOption = ConnectionOption::Default);
|
||||
|
||||
/**
|
||||
* @name Packet request functions
|
||||
|
||||
+185
-41
@@ -5,6 +5,7 @@
|
||||
#include "Common/FDUtils.h"
|
||||
#include "FEXCore/Utils/Allocator.h"
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Tests/VDSO_Emulation.h"
|
||||
#include "Linux/Utils/ELFParser.h"
|
||||
#include "Linux/Utils/ELFSymbolDatabase.h"
|
||||
|
||||
@@ -14,12 +15,14 @@
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <list>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
@@ -177,19 +180,21 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
|
||||
static std::string ResolveRootfsFile(std::string const &File, std::string RootFS) {
|
||||
// If the path is relative then just run that
|
||||
if (std::filesystem::path(File).is_relative()) {
|
||||
if (File[0] != '/') {
|
||||
return File;
|
||||
}
|
||||
|
||||
std::string RootFSLink = RootFS + File;
|
||||
|
||||
while (std::filesystem::is_symlink(RootFSLink)) {
|
||||
char Filename[PATH_MAX];
|
||||
while(FEX::HLE::IsSymlink(RootFSLink)) {
|
||||
// Do some special handling if the RootFS's linker is a symlink
|
||||
// Ubuntu's rootFS by default provides an absolute location symlink to the linker
|
||||
// Resolve this around back to the rootfs
|
||||
auto SymlinkTarget = std::filesystem::read_symlink(RootFSLink);
|
||||
if (SymlinkTarget.is_absolute()) {
|
||||
RootFSLink = RootFS + SymlinkTarget.string();
|
||||
auto SymlinkSize = FEX::HLE::GetSymlink(RootFSLink, Filename, PATH_MAX - 1);
|
||||
if (SymlinkSize > 0 && Filename[0] == '/') {
|
||||
RootFSLink = RootFS;
|
||||
RootFSLink += std::string_view(Filename, SymlinkSize);
|
||||
}
|
||||
else {
|
||||
break;
|
||||
@@ -353,7 +358,36 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
//
|
||||
// This is still technically a memory leak if the stack grows, but since the primary thread's stack only gets destroyed on process close, this is
|
||||
// fine.
|
||||
StackPointer = reinterpret_cast<uintptr_t>(Mapper(nullptr, StackSize(), PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
|
||||
|
||||
// Stacks need to be allocated at the hint location just like on a real x86 system.
|
||||
// These are 128MB regions on both x86-64 and x86.
|
||||
//
|
||||
// These are required to be in the correct location taking up the appropriate 128MB of space, otherwise the wine preloader crashes FEX.
|
||||
// This is due to the wine-preloader hardcoding addresses [0x7FFFFE000000 - 0x7FFFFFFF0000) as a top-down
|
||||
// allocation region. They use mmap with MAP_FIXED, ignoring any previously mapped area at that location and overwriting it.
|
||||
// Wine-preloader is expecting to allocate 32MB out of the total 128MB stack space in this case. Leaving 96MB for the application.
|
||||
//
|
||||
// If FEX doesn't allocate the stack in this region (nullptr mmap hint) then later allocations that FEX does will /eventually/
|
||||
// end up inside of this address space that wine allocates. This usually ends up being a JIT CodeBuffer, which zeroes the memory and faults with a
|
||||
// SIGILL.
|
||||
//
|
||||
// On the upside, this more accurately emulates how the kernel allocates stack space for the application when hinting at the location.
|
||||
//
|
||||
void* StackPointerBase{};
|
||||
uint64_t StackHint = Is64BitMode() ? STACK_HINT_64 : STACK_HINT_32;
|
||||
|
||||
// Allocate the base of the full 128MB stack range.
|
||||
StackPointerBase = Mapper(reinterpret_cast<void*>(StackHint), FULL_STACK_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN | MAP_NORESERVE, -1, 0);
|
||||
|
||||
if (StackPointerBase == reinterpret_cast<void*>(~0ULL)) {
|
||||
LogMan::Msg::EFmt("Allocating stack failed");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Allocate with permissions the 8MB of regular stack size.
|
||||
StackPointer = reinterpret_cast<uintptr_t>(Mapper(
|
||||
reinterpret_cast<void*>(reinterpret_cast<uint64_t>(StackPointerBase) + FULL_STACK_SIZE - StackSize()),
|
||||
StackSize(), PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
|
||||
|
||||
if (StackPointer == ~0ULL) {
|
||||
LogMan::Msg::EFmt("Allocating stack failed");
|
||||
@@ -499,28 +533,36 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
AuxVariables.emplace_back(auxv_t{14, getauxval(AT_EGID)}); // AT_EGID
|
||||
AuxVariables.emplace_back(auxv_t{17, getauxval(AT_CLKTCK)}); // AT_CLKTIK
|
||||
AuxVariables.emplace_back(auxv_t{6, 0x1000}); // AT_PAGESIZE
|
||||
AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
|
||||
AuxVariables.emplace_back(auxv_t{23, 0}); // AT_SECURE
|
||||
AuxRandom = &AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
|
||||
AuxVariables.emplace_back(auxv_t{23, getauxval(AT_SECURE)}); // AT_SECURE
|
||||
AuxVariables.emplace_back(auxv_t{8, 0}); // AT_FLAGS
|
||||
AuxVariables.emplace_back(auxv_t{5, MainElf.phdrs.size()}); // AT_PHNUM
|
||||
AuxVariables.emplace_back(auxv_t{16, HWCap}); // AT_HWCAP
|
||||
AuxVariables.emplace_back(auxv_t{26, HWCap2}); // AT_HWCAP2
|
||||
AuxVariables.emplace_back(auxv_t{51, CalculateSignalStackSize()}); // AT_MINSIGSTKSZ
|
||||
AuxPlatform = &AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
|
||||
|
||||
if (Is64BitMode()) {
|
||||
AuxVariables.emplace_back(auxv_t{4, 0x38}); // AT_PHENT
|
||||
// On x86 this is the value returned from CPUID 01h EDX
|
||||
AuxVariables.emplace_back(auxv_t{16, 0}); // AT_HWCAP
|
||||
|
||||
//AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
|
||||
// On x86 only allows userspace to check for monitor and fs/gs base writing in CPL3
|
||||
//AuxVariables.emplace_back(auxv_t{26, 0}); // AT_HWCAP2
|
||||
|
||||
// we don't support vsyscall so we don't set those
|
||||
//AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
|
||||
}
|
||||
else {
|
||||
AuxVariables.emplace_back(auxv_t{4, 0x20}); // AT_PHENT
|
||||
|
||||
// we don't support vsyscall so we don't set those
|
||||
//AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
|
||||
auto VSyscallEntry = FEX::VDSO::GetVSyscallEntry(VDSOBase);
|
||||
if (!VSyscallEntry) [[unlikely]] {
|
||||
// If the VDSO thunk doesn't exist then we might not have a vsyscall entry.
|
||||
// Newer glibc requires vsyscall to exist now. So let's allocate a buffer and stick a vsyscall in to it.
|
||||
auto VSyscallPage = Mapper(nullptr, FHU::FEX_PAGE_SIZE, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
constexpr static uint8_t VSyscallCode[] = {
|
||||
0xcd, 0x80, // int 0x80
|
||||
0xc3, // ret
|
||||
};
|
||||
memcpy(VSyscallPage, VSyscallCode, sizeof(VSyscallCode));
|
||||
mprotect(VSyscallPage, FHU::FEX_PAGE_SIZE, PROT_READ);
|
||||
VSyscallEntry = reinterpret_cast<uint64_t>(VSyscallPage);
|
||||
}
|
||||
|
||||
AuxVariables.emplace_back(auxv_t{32, VSyscallEntry}); // AT_SYSINFO - Entry point to syscall
|
||||
}
|
||||
|
||||
if (VDSOBase) {
|
||||
@@ -550,10 +592,11 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
uint64_t EnvpOffset,
|
||||
const std::vector<std::string> &Args,
|
||||
const std::vector<std::string> &EnvironmentVariables,
|
||||
const std::vector<auxv_t> &AuxVariables,
|
||||
const std::list<auxv_t> &AuxVariables,
|
||||
uint64_t *AuxTabBase,
|
||||
uint64_t *AuxTabSize,
|
||||
PointerType RandomNumberOffset
|
||||
PointerType RandomNumberOffset,
|
||||
PointerType PlatformNameOffset
|
||||
) {
|
||||
// Pointer list offsets
|
||||
PointerType *ArgumentPointers = reinterpret_cast<PointerType*>(StackPointer + PointerSize);
|
||||
@@ -607,20 +650,10 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
// Last envp needs to be nullptr
|
||||
EnvpPointers[EnvironmentVariables.size()] = 0;
|
||||
|
||||
for (size_t i = 0; i < AuxVariables.size(); ++i) {
|
||||
if (AuxVariables[i].key == 25) {
|
||||
// Random value is always 128bits
|
||||
AuxType Random{25, static_cast<PointerType>(StackPointer + RandomNumberOffset)};
|
||||
uint64_t *RandomLoc = reinterpret_cast<uint64_t*>(StackPointer + RandomNumberOffset);
|
||||
RandomLoc[0] = 0xDEAD;
|
||||
RandomLoc[1] = 0xDEAD2;
|
||||
AuxVPointers[i].key = Random.key;
|
||||
AuxVPointers[i].val = Random.val;
|
||||
}
|
||||
else {
|
||||
AuxVPointers[i].key = AuxVariables[i].key;
|
||||
AuxVPointers[i].val = AuxVariables[i].val;
|
||||
}
|
||||
for (size_t i = 0; auto const &Variable : AuxVariables) {
|
||||
AuxVPointers[i].key = Variable.key;
|
||||
AuxVPointers[i].val = Variable.val;
|
||||
++i;
|
||||
}
|
||||
|
||||
*AuxTabBase = reinterpret_cast<uint64_t>(AuxVPointers);
|
||||
@@ -656,12 +689,44 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
TotalArgumentMemSize += EnvironmentBackingSize;
|
||||
|
||||
// Random number location
|
||||
uint32_t RandomNumberLocation = TotalArgumentMemSize;
|
||||
uint64_t RandomNumberLocation = TotalArgumentMemSize;
|
||||
TotalArgumentMemSize += 16;
|
||||
|
||||
uint64_t PlatformNameLocation = TotalArgumentMemSize;
|
||||
TotalArgumentMemSize += platform_string_max_size;
|
||||
|
||||
// Offset the stack by how much memory we need
|
||||
StackPointer -= TotalArgumentMemSize;
|
||||
|
||||
// Setup our AUXP values that need memory now that the stack is setup
|
||||
AuxPlatform->val = StackPointer + PlatformNameLocation;
|
||||
char *PlatformLoc = reinterpret_cast<char*>(AuxPlatform->val);
|
||||
memset(PlatformLoc, 0, platform_string_max_size);
|
||||
if (Is64BitMode()) {
|
||||
strncpy(PlatformLoc, platform_name_x86_64.data(), platform_string_max_size);
|
||||
}
|
||||
else {
|
||||
strncpy(PlatformLoc, platform_name_i686.data(), platform_string_max_size);
|
||||
}
|
||||
|
||||
// Random value is always 128bits
|
||||
AuxRandom->val = StackPointer + RandomNumberLocation;
|
||||
uint64_t *RandomLoc = reinterpret_cast<uint64_t*>(AuxRandom->val);
|
||||
uint64_t *HostRandom = reinterpret_cast<uint64_t*>(getauxval(AT_RANDOM));
|
||||
if (HostRandom) {
|
||||
// Pass through the host's random values
|
||||
RandomLoc[0] = HostRandom[0];
|
||||
RandomLoc[1] = HostRandom[1];
|
||||
}
|
||||
else {
|
||||
// Nothing provided from the kernel, generate our own random values.
|
||||
std::random_device rd;
|
||||
std::uniform_int_distribution<uint64_t> d(0);
|
||||
|
||||
RandomLoc[0] = d(rd);
|
||||
RandomLoc[1] = d(rd);
|
||||
}
|
||||
|
||||
// Stack setup
|
||||
// [0, 8): Argument Count
|
||||
// [8, 16): Argument Pointer 0
|
||||
@@ -687,7 +752,8 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
AuxVariables,
|
||||
&AuxTabBase,
|
||||
&AuxTabSize,
|
||||
RandomNumberLocation
|
||||
RandomNumberLocation,
|
||||
PlatformNameLocation
|
||||
);
|
||||
}
|
||||
else {
|
||||
@@ -701,7 +767,8 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
AuxVariables,
|
||||
&AuxTabBase,
|
||||
&AuxTabSize,
|
||||
RandomNumberLocation
|
||||
RandomNumberLocation,
|
||||
PlatformNameLocation
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -718,7 +785,7 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
return BaseOffset;
|
||||
}
|
||||
|
||||
bool Is64BitMode() {
|
||||
bool Is64BitMode() const {
|
||||
return MainElf.type == ::ELFLoader::ELFContainer::TYPE_X86_64;
|
||||
}
|
||||
|
||||
@@ -734,19 +801,96 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
VDSOBase = Base;
|
||||
}
|
||||
|
||||
void CalculateHWCaps(FEXCore::Context::Context *ctx) {
|
||||
// HWCAP is just CPUID function 0x1, the EDX result
|
||||
auto res_1 = FEXCore::Context::RunCPUIDFunction(ctx, 1, 0);
|
||||
HWCap = res_1.edx;
|
||||
|
||||
// HWCAP2 is as follows:
|
||||
// Bits:
|
||||
// 0 - MONITOR/MWAIT available in CPL3
|
||||
// 1 - FSGSBASE instructions available in CPL3
|
||||
HWCap2 = 0;
|
||||
|
||||
// We need to know if we support AVX for AT_MINSIGSTKSZ
|
||||
SupportsAVX = !!(res_1.ecx & (1U << 28));
|
||||
}
|
||||
|
||||
uint64_t CalculateSignalStackSize() const {
|
||||
// We must calculate the required signal stack size that the "kernel" consumes.
|
||||
// For FEX this means the amount of state we store in to the guest stack, not including the amount
|
||||
// that FEX stores in to the host stack as well.
|
||||
//
|
||||
// This needs to match what we do in FEXCore's dispatcher (Which should at some point be moved to the frontend).
|
||||
//
|
||||
// This roughly means that we need to calculate the combined size of:
|
||||
// - xstate or _libc_fstate depending on AVX support
|
||||
// - ucontext_t
|
||||
// - siginfo_t
|
||||
// Size of state requiring to be stored is different between 32-bit and 64-bit.
|
||||
|
||||
uint64_t Result{};
|
||||
if (Is64BitMode()) {
|
||||
Result += sizeof(FEXCore::x86_64::ucontext_t);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86_64::ucontext_t));
|
||||
if (SupportsAVX) {
|
||||
Result += sizeof(FEXCore::x86_64::xstate);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86_64::xstate));
|
||||
}
|
||||
else {
|
||||
Result += sizeof(FEXCore::x86_64::_libc_fpstate);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86_64::_libc_fpstate));
|
||||
}
|
||||
|
||||
Result += sizeof(siginfo_t);
|
||||
Result = FEXCore::AlignUp(Result, alignof(siginfo_t));
|
||||
}
|
||||
else {
|
||||
Result += sizeof(FEXCore::x86::ucontext_t);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86::ucontext_t));
|
||||
if (SupportsAVX) {
|
||||
Result += sizeof(FEXCore::x86::xstate);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86::xstate));
|
||||
}
|
||||
else {
|
||||
Result += sizeof(FEXCore::x86::_libc_fpstate);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86::_libc_fpstate));
|
||||
}
|
||||
|
||||
Result += sizeof(FEXCore::x86::siginfo_t);
|
||||
Result = FEXCore::AlignUp(Result, alignof(FEXCore::x86::siginfo_t));
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
constexpr static uint64_t BRK_SIZE = 8 * 1024 * 1024;
|
||||
constexpr static uint64_t STACK_SIZE = 8 * 1024 * 1024;
|
||||
constexpr static uint64_t FULL_STACK_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static uint64_t STACK_HINT_32 = 0xFFFFE000 - FULL_STACK_SIZE;
|
||||
constexpr static uint64_t STACK_HINT_64 = 0x7FFFFFFFF000 - FULL_STACK_SIZE;
|
||||
|
||||
std::vector<std::string> Args;
|
||||
std::vector<std::string> EnvironmentVariables;
|
||||
std::vector<char const*> LoaderArgs;
|
||||
|
||||
std::vector<auxv_t> AuxVariables;
|
||||
std::list<auxv_t> AuxVariables;
|
||||
uint64_t AuxTabBase, AuxTabSize;
|
||||
uint64_t ArgumentBackingSize{};
|
||||
uint64_t EnvironmentBackingSize{};
|
||||
uint64_t BaseOffset{};
|
||||
void* VDSOBase{};
|
||||
uint64_t HWCap{};
|
||||
uint64_t HWCap2{};
|
||||
bool SupportsAVX{};
|
||||
|
||||
auxv_t *AuxRandom{};
|
||||
auxv_t *AuxPlatform{};
|
||||
|
||||
static constexpr std::string_view platform_name_x86_64 = "x86_64";
|
||||
static constexpr std::string_view platform_name_i686 = "i686";
|
||||
// Need to include null character.
|
||||
static constexpr size_t platform_string_max_size = std::max(platform_name_x86_64.size(), platform_name_i686.size()) + 1;
|
||||
|
||||
FEX_CONFIG_OPT(AdditionalArguments, ADDITIONALARGUMENTS);
|
||||
};
|
||||
+51
-45
@@ -24,6 +24,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cerrno>
|
||||
@@ -190,10 +191,9 @@ bool IsInterpreterInstalled() {
|
||||
// The interpreter is installed if both the binfmt_misc handlers are available
|
||||
// Or if we were originally executed with FD. Which means the interpreter is installed
|
||||
|
||||
std::error_code ec{};
|
||||
return ExecutedWithFD ||
|
||||
(std::filesystem::exists("/proc/sys/fs/binfmt_misc/FEX-x86", ec) &&
|
||||
std::filesystem::exists("/proc/sys/fs/binfmt_misc/FEX-x86_64", ec));
|
||||
(access("/proc/sys/fs/binfmt_misc/FEX-x86", F_OK) == 0 &&
|
||||
access("/proc/sys/fs/binfmt_misc/FEX-x86_64", F_OK) == 0);
|
||||
}
|
||||
|
||||
int main(int argc, char **argv, char **const envp) {
|
||||
@@ -283,6 +283,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Profiler::Init();
|
||||
FEXCore::Telemetry::Initialize();
|
||||
|
||||
RootFSRedirect(&Program.first, LDPath());
|
||||
@@ -393,6 +394,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
// Load VDSO in to memory prior to mapping our ELFs.
|
||||
void* VDSOBase = FEX::VDSO::LoadVDSOThunks(Loader.Is64BitMode(), Mapper);
|
||||
Loader.SetVDSOBase(VDSOBase);
|
||||
Loader.CalculateHWCaps(CTX);
|
||||
|
||||
if (!Loader.MapMemory(Mapper, Unmapper)) {
|
||||
// failed to map
|
||||
@@ -425,38 +427,39 @@ int main(int argc, char **argv, char **const envp) {
|
||||
});
|
||||
}
|
||||
|
||||
if (AOTIRLoad() || AOTIRCapture() || AOTIRGenerate()) {
|
||||
const bool AOTEnabled = AOTIRLoad() || AOTIRCapture() || AOTIRGenerate();
|
||||
if (AOTEnabled) {
|
||||
LogMan::Msg::IFmt("Warning: AOTIR is experimental, and might lead to crashes. "
|
||||
"Capture doesn't work with programs that fork.");
|
||||
|
||||
FEXCore::Context::SetAOTIRLoader(CTX, [](const std::string &fileid) -> int {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir");
|
||||
|
||||
return open(filepath.c_str(), O_RDONLY);
|
||||
});
|
||||
|
||||
FEXCore::Context::SetAOTIRWriter(CTX, [](const std::string& fileid) -> std::unique_ptr<std::ofstream> {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir.tmp");
|
||||
auto AOTWrite = std::make_unique<std::ofstream>(filepath, std::ios::out | std::ios::binary);
|
||||
if (*AOTWrite) {
|
||||
std::filesystem::resize_file(filepath, 0);
|
||||
AOTWrite->seekp(0);
|
||||
LogMan::Msg::IFmt("AOTIR: Storing {}", fileid);
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: Failed to store {}", fileid);
|
||||
}
|
||||
return AOTWrite;
|
||||
});
|
||||
|
||||
FEXCore::Context::SetAOTIRRenamer(CTX, [](const std::string& fileid) -> void {
|
||||
auto TmpFilepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir.tmp");
|
||||
auto NewFilepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir");
|
||||
|
||||
// Rename the temporary file to atomically update the file
|
||||
std::filesystem::rename(TmpFilepath, NewFilepath);
|
||||
});
|
||||
}
|
||||
|
||||
FEXCore::Context::SetAOTIRLoader(CTX, [](const std::string &fileid) -> int {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir");
|
||||
|
||||
return open(filepath.c_str(), O_RDONLY);
|
||||
});
|
||||
|
||||
FEXCore::Context::SetAOTIRWriter(CTX, [](const std::string& fileid) -> std::unique_ptr<std::ofstream> {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir.tmp");
|
||||
auto AOTWrite = std::make_unique<std::ofstream>(filepath, std::ios::out | std::ios::binary);
|
||||
if (*AOTWrite) {
|
||||
std::filesystem::resize_file(filepath, 0);
|
||||
AOTWrite->seekp(0);
|
||||
LogMan::Msg::IFmt("AOTIR: Storing {}", fileid);
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: Failed to store {}", fileid);
|
||||
}
|
||||
return AOTWrite;
|
||||
});
|
||||
|
||||
FEXCore::Context::SetAOTIRRenamer(CTX, [](const std::string& fileid) -> void {
|
||||
auto TmpFilepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir.tmp");
|
||||
auto NewFilepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".aotir");
|
||||
|
||||
// Rename the temporary file to atomically update the file
|
||||
std::filesystem::rename(TmpFilepath, NewFilepath);
|
||||
});
|
||||
|
||||
if (AOTIRGenerate()) {
|
||||
for(auto &Section: Loader.Sections) {
|
||||
FEX::AOT::AOTGenSection(CTX, Section);
|
||||
@@ -465,21 +468,23 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Context::RunUntilExit(CTX);
|
||||
}
|
||||
|
||||
std::filesystem::create_directories(std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir", ec);
|
||||
if (!ec) {
|
||||
FEXCore::Context::WriteFilesWithCode(CTX, [](const std::string& fileid, const std::string& filename) {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".path");
|
||||
int fd = open(filepath.c_str(), O_CREAT | O_EXCL | O_WRONLY, 0644);
|
||||
if (fd != -1) {
|
||||
write(fd, filename.c_str(), filename.size());
|
||||
close(fd);
|
||||
}
|
||||
});
|
||||
}
|
||||
if (AOTEnabled) {
|
||||
std::filesystem::create_directories(std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir", ec);
|
||||
if (!ec) {
|
||||
FEXCore::Context::WriteFilesWithCode(CTX, [](const std::string& fileid, const std::string& filename) {
|
||||
auto filepath = std::filesystem::path(FEXCore::Config::GetDataDirectory()) / "aotir" / (fileid + ".path");
|
||||
int fd = open(filepath.c_str(), O_CREAT | O_EXCL | O_WRONLY, 0644);
|
||||
if (fd != -1) {
|
||||
write(fd, filename.c_str(), filename.size());
|
||||
close(fd);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
if (AOTIRCapture() || AOTIRGenerate()) {
|
||||
FEXCore::Context::FinalizeAOTIRCache(CTX);
|
||||
LogMan::Msg::IFmt("AOTIR Cache Stored");
|
||||
if (AOTIRCapture() || AOTIRGenerate()) {
|
||||
FEXCore::Context::FinalizeAOTIRCache(CTX);
|
||||
LogMan::Msg::IFmt("AOTIR Cache Stored");
|
||||
}
|
||||
}
|
||||
|
||||
auto ProgramStatus = FEXCore::Context::GetProgramStatus(CTX);
|
||||
@@ -502,6 +507,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Allocator::ReclaimMemoryRegion(Base48Bit);
|
||||
// Allocator is now original system allocator
|
||||
FEXCore::Telemetry::Shutdown(Program.second);
|
||||
FEXCore::Profiler::Shutdown();
|
||||
if (ShutdownReason == FEXCore::Context::ExitReason::EXIT_SHUTDOWN) {
|
||||
return ProgramStatus;
|
||||
}
|
||||
|
||||
@@ -127,13 +127,13 @@ namespace FEX::HarnessHelper {
|
||||
|
||||
// GS
|
||||
if (MatchMask & 1) {
|
||||
CheckGPRs("GS", State1.gs, State2.gs);
|
||||
CheckGPRs("GS", State1.gs_cached, State2.gs_cached);
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
|
||||
// FS
|
||||
if (MatchMask & 1) {
|
||||
CheckGPRs("FS", State1.fs, State2.fs);
|
||||
CheckGPRs("FS", State1.fs_cached, State2.fs_cached);
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
|
||||
@@ -233,8 +233,8 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[13][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[14][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
@@ -280,8 +280,8 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[13][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[14][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
|
||||
@@ -622,6 +622,11 @@ namespace FEX::EmulatedFile {
|
||||
EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx} {
|
||||
FDReadCreators["/proc/cpuinfo"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
// Only allow a single thread to initialize the cpu_info.
|
||||
// Jit in-case multiple threads try to initialize at once.
|
||||
// Check if deferred cpuinfo initialization has occured.
|
||||
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig()); });
|
||||
|
||||
int FD = GenTmpFD();
|
||||
write(FD, (void*)&cpu_info.at(0), cpu_info.size());
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
@@ -700,8 +705,6 @@ namespace FEX::EmulatedFile {
|
||||
if (CPUCores > 1) {
|
||||
cpus_online += "-" + std::to_string(CPUCores - 1);
|
||||
}
|
||||
|
||||
cpu_info = GenerateCPUInfo(ctx, CPUCores);
|
||||
}
|
||||
|
||||
EmulatedFDManager::~EmulatedFDManager() {
|
||||
@@ -734,7 +737,7 @@ namespace FEX::EmulatedFile {
|
||||
}
|
||||
|
||||
std::error_code ec;
|
||||
bool exists = std::filesystem::exists(Path, ec);
|
||||
bool exists = access(Path.c_str(), F_OK) == 0;
|
||||
if (ec) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
@@ -27,6 +27,7 @@ namespace FEX::EmulatedFile {
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
std::string cpus_online{};
|
||||
std::once_flag cpu_info_initialized{};
|
||||
std::string cpu_info{};
|
||||
using FDReadStringFunc = std::function<int32_t(FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode)>;
|
||||
std::unordered_map<std::string, FDReadStringFunc> FDReadCreators;
|
||||
|
||||
@@ -230,7 +230,7 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
auto LoadThunksDB = [this, ThunkGuestPath](bool *LoadedThunkDatabase, json_t const* ThunksDB) {
|
||||
// If a thunks DB property exists then we pull in data from the thunks database
|
||||
// Load the initial thunks database
|
||||
if (LoadedThunkDatabase) {
|
||||
if (!*LoadedThunkDatabase) {
|
||||
LoadThunkDatabase(true);
|
||||
LoadThunkDatabase(false);
|
||||
*LoadedThunkDatabase = true;
|
||||
@@ -239,62 +239,50 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
// Now load this property
|
||||
for (json_t const* Item = json_getChild(ThunksDB); Item != nullptr; Item = json_getSibling(Item)) {
|
||||
const char *LibraryName = json_getName(Item);
|
||||
int64_t LibraryEnabled = json_getInteger(Item);
|
||||
if (LibraryEnabled != 0) {
|
||||
// If the library is enabled then find it in the DB
|
||||
// Enable the overlay and all the dependencies in one go
|
||||
auto DBObject = ThunkDB.find(LibraryName);
|
||||
if (DBObject != ThunkDB.end() &&
|
||||
DBObject->second.Enabled == false) {
|
||||
|
||||
auto ThunkPath = ThunkGuestPath / DBObject->second.LibraryName;
|
||||
if (std::filesystem::exists(ThunkPath)) {
|
||||
for (auto Overlay : DBObject->second.Overlays) {
|
||||
// Direct full path in guest RootFS to our overlay file
|
||||
ThunkOverlays.emplace(Overlay, ThunkPath);
|
||||
}
|
||||
}
|
||||
DBObject->second.Enabled = true;
|
||||
// Now walk the dependencies and set them up as well
|
||||
// Make sure to enable each one as we go to remove circular dependencies
|
||||
std::function<void(std::unordered_set<std::string> &Depends)> InsertDependencies
|
||||
= [this, &ThunkGuestPath, &InsertDependencies](std::unordered_set<std::string> &Depends) -> void {
|
||||
for (auto &Depend : Depends) {
|
||||
auto DBDepend = ThunkDB.find(Depend);
|
||||
if (DBDepend != ThunkDB.end() &&
|
||||
DBDepend->second.Enabled == false) {
|
||||
|
||||
auto ThunkPath = ThunkGuestPath / DBDepend->second.LibraryName;
|
||||
if (std::filesystem::exists(ThunkPath)) {
|
||||
for (auto Overlay : DBDepend->second.Overlays) {
|
||||
// Direct full path in guest RootFS to our overlay file
|
||||
ThunkOverlays.emplace(Overlay, ThunkPath);
|
||||
}
|
||||
}
|
||||
|
||||
// Enabled, now walk this dependencies
|
||||
DBDepend->second.Enabled = true;
|
||||
InsertDependencies(DBDepend->second.Depends);
|
||||
}
|
||||
}
|
||||
};
|
||||
InsertDependencies(DBObject->second.Depends);
|
||||
}
|
||||
bool LibraryEnabled = json_getInteger(Item) != 0;
|
||||
// If the library is enabled then find it in the DB
|
||||
// Enable the overlay and all the dependencies in one go
|
||||
auto DBObject = ThunkDB.find(LibraryName);
|
||||
if (DBObject != ThunkDB.end()) {
|
||||
DBObject->second.Enabled = LibraryEnabled;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// We try to load ThunksDB from {FEX global config, FEX user config, AppConfig Global, AppConfig Local, Defined ThunksConfig option}
|
||||
// We try to load ThunksDB from:
|
||||
// - FEX global config
|
||||
// - FEX user config
|
||||
// - Defined ThunksConfig option
|
||||
// - Steam AppConfig Global
|
||||
// - AppConfig Global
|
||||
// - Steam AppConfig Local
|
||||
// - AppConfig Local
|
||||
// This doesn't support the classic thunks interface.
|
||||
|
||||
auto AppName = AppConfigName();
|
||||
std::vector<std::string> ConfigPaths {
|
||||
FEXCore::Config::GetConfigFileLocation(true),
|
||||
FEXCore::Config::GetConfigFileLocation(false),
|
||||
FEXCore::Config::GetApplicationConfig(AppConfigName(), true),
|
||||
FEXCore::Config::GetApplicationConfig(AppConfigName(), false),
|
||||
ThunkConfigFile,
|
||||
};
|
||||
|
||||
auto SteamID = getenv("SteamAppId");
|
||||
if (SteamID) {
|
||||
// If a SteamID exists then let's search for Steam application configs as well.
|
||||
// We want to key off both the SteamAppId number /and/ the executable since we may not want to thunk all binaries.
|
||||
auto SteamAppName = fmt::format("Steam_{}_{}", SteamID, AppName);
|
||||
|
||||
// Steam application configs interleaved with non-steam for priority sorting.
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(SteamAppName, true));
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(AppName, true));
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(SteamAppName, false));
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(AppName, false));
|
||||
}
|
||||
else {
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(AppName, true));
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(AppName, false));
|
||||
}
|
||||
|
||||
for (const auto &Path : ConfigPaths) {
|
||||
std::vector<char> FileData;
|
||||
if (LoadFile(FileData, Path)) {
|
||||
@@ -313,6 +301,40 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
}
|
||||
}
|
||||
|
||||
// Now that we loaded the thunks object, walk through and ensure dependencies are enabled as well.
|
||||
for (auto const &DBObject : ThunkDB) {
|
||||
if (!DBObject.second.Enabled) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Now walk the dependencies and set them up as well
|
||||
// Make sure to enable each one as we go to remove circular dependencies
|
||||
std::function<void(const std::unordered_set<std::string> &Depends, bool AlreadyEnabled)> InsertDependencies
|
||||
= [this, &ThunkGuestPath, &InsertDependencies](const std::unordered_set<std::string> &Depends, bool AlreadyEnabled) -> void {
|
||||
for (auto const &Depend : Depends) {
|
||||
auto DBDepend = ThunkDB.find(Depend);
|
||||
if (DBDepend != ThunkDB.end() &&
|
||||
(DBDepend->second.Enabled == false || AlreadyEnabled)) {
|
||||
|
||||
auto ThunkPath = ThunkGuestPath / DBDepend->second.LibraryName;
|
||||
if (std::filesystem::exists(ThunkPath)) {
|
||||
for (const auto& Overlay : DBDepend->second.Overlays) {
|
||||
// Direct full path in guest RootFS to our overlay file
|
||||
ThunkOverlays.emplace(Overlay, ThunkPath);
|
||||
}
|
||||
}
|
||||
|
||||
// Enabled, now walk this dependencies
|
||||
DBDepend->second.Enabled = true;
|
||||
InsertDependencies(DBDepend->second.Depends, false);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
InsertDependencies({DBObject.first}, true);
|
||||
InsertDependencies(DBObject.second.Depends, false);
|
||||
}
|
||||
|
||||
// Now clear the thunk database since we're loaded
|
||||
ThunkDB.clear();
|
||||
|
||||
@@ -333,7 +355,6 @@ FileManager::~FileManager() {
|
||||
}
|
||||
|
||||
std::string FileManager::GetEmulatedPath(const char *pathname, bool FollowSymlink) {
|
||||
auto RootFSPath = LDPath();
|
||||
if (!pathname || // If no pathname
|
||||
pathname[0] != '/' || // If relative
|
||||
strcmp(pathname, "/") == 0) { // If we are getting root
|
||||
@@ -345,17 +366,19 @@ std::string FileManager::GetEmulatedPath(const char *pathname, bool FollowSymlin
|
||||
return thunkOverlay->second;
|
||||
}
|
||||
|
||||
auto RootFSPath = LDPath();
|
||||
if (RootFSPath.empty()) { // If RootFS doesn't exist
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string Path = RootFSPath + pathname;
|
||||
if (FollowSymlink) {
|
||||
std::error_code ec;
|
||||
while(std::filesystem::is_symlink(Path, ec)) {
|
||||
auto SymlinkTarget = std::filesystem::read_symlink(Path);
|
||||
if (SymlinkTarget.is_absolute()) {
|
||||
Path = RootFSPath + SymlinkTarget.string();
|
||||
char Filename[PATH_MAX];
|
||||
while(FEX::HLE::IsSymlink(Path)) {
|
||||
auto SymlinkSize = FEX::HLE::GetSymlink(Path, Filename, PATH_MAX - 1);
|
||||
if (SymlinkSize > 0 && Filename[0] == '/') {
|
||||
Path = RootFSPath;
|
||||
Path += std::string_view(Filename, SymlinkSize);
|
||||
}
|
||||
else {
|
||||
break;
|
||||
|
||||
@@ -15,6 +15,7 @@ $end_info$
|
||||
#include <stddef.h>
|
||||
#include <string>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
|
||||
#include <unordered_map>
|
||||
@@ -27,6 +28,18 @@ struct Context;
|
||||
}
|
||||
|
||||
namespace FEX::HLE {
|
||||
[[maybe_unused]]
|
||||
static bool IsSymlink(const std::string &Filename) {
|
||||
// Checks to see if a filepath is a symlink.
|
||||
struct stat Buffer{};
|
||||
int Result = lstat(Filename.c_str(), &Buffer);
|
||||
return Result == 0 && S_ISLNK(Buffer.st_mode);
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static ssize_t GetSymlink(const std::string &Filename, char *ResultBuffer, size_t ResultBufferSize) {
|
||||
return readlink(Filename.c_str(), ResultBuffer, ResultBufferSize);
|
||||
}
|
||||
|
||||
struct open_how;
|
||||
|
||||
|
||||
@@ -455,7 +455,7 @@ namespace FEX::HLE {
|
||||
// Ignore a non-canonical address
|
||||
return -EPERM;
|
||||
}
|
||||
Frame->State.gs = addr;
|
||||
Frame->State.gs_cached = addr;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x1002: // ARCH_SET_FS
|
||||
@@ -463,15 +463,15 @@ namespace FEX::HLE {
|
||||
// Ignore a non-canonical address
|
||||
return -EPERM;
|
||||
}
|
||||
Frame->State.fs = addr;
|
||||
Frame->State.fs_cached = addr;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x1003: // ARCH_GET_FS
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.fs;
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.fs_cached;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x1004: // ARCH_GET_GS
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.gs;
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.gs_cached;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x3001: // ARCH_CET_STATUS
|
||||
|
||||
@@ -15,6 +15,7 @@ extern "C" {
|
||||
#include "fex-drm/panfrost_drm.h"
|
||||
#include "fex-drm/msm_drm.h"
|
||||
#include "fex-drm/nouveau_drm.h"
|
||||
#include "fex-drm/radeon_drm.h"
|
||||
#include "fex-drm/vc4_drm.h"
|
||||
#include "fex-drm/v3d_drm.h"
|
||||
#include "fex-drm/virtgpu_drm.h"
|
||||
@@ -713,6 +714,360 @@ fex_drm_amdgpu_gem_metadata {
|
||||
};
|
||||
}
|
||||
|
||||
namespace RADEON {
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_gem_create")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_gem_create {
|
||||
compat_uint64_t size;
|
||||
compat_uint64_t alignment;
|
||||
__u32 handle;
|
||||
__u32 initial_domain;
|
||||
__u32 flags;
|
||||
|
||||
fex_drm_radeon_gem_create() = delete;
|
||||
|
||||
operator drm_radeon_gem_create() const {
|
||||
drm_radeon_gem_create val{};
|
||||
val.size = size;
|
||||
val.alignment = alignment;
|
||||
val.handle = handle;
|
||||
val.initial_domain = initial_domain;
|
||||
val.flags = flags;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_gem_create(struct drm_radeon_gem_create val) {
|
||||
size = val.size;
|
||||
alignment = val.alignment;
|
||||
handle = val.handle;
|
||||
initial_domain = val.initial_domain;
|
||||
flags = val.flags;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_PACKED
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_init")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_init_t {
|
||||
enum {
|
||||
} func;
|
||||
|
||||
compat_ulong_t sarea_priv_offset;
|
||||
int32_t is_pci;
|
||||
int32_t cp_mode;
|
||||
int32_t gart_size;
|
||||
int32_t ring_size;
|
||||
int32_t usec_timeout;
|
||||
|
||||
uint32_t fb_bpp;
|
||||
uint32_t front_offset, front_pitch;
|
||||
uint32_t back_offset, back_pitch;
|
||||
uint32_t depth_bpp;
|
||||
uint32_t depth_offset, depth_pitch;
|
||||
|
||||
compat_ulong_t fb_offset;
|
||||
compat_ulong_t mmio_offset;
|
||||
compat_ulong_t ring_offset;
|
||||
compat_ulong_t ring_rptr_offset;
|
||||
compat_ulong_t buffers_offset;
|
||||
compat_ulong_t gart_textures_offset;
|
||||
|
||||
fex_drm_radeon_init_t() = delete;
|
||||
|
||||
operator drm_radeon_init_t() const {
|
||||
drm_radeon_init_t val{};
|
||||
val.sarea_priv_offset = sarea_priv_offset;
|
||||
val.is_pci = is_pci;
|
||||
val.cp_mode = cp_mode;
|
||||
val.gart_size = gart_size;
|
||||
val.ring_size = ring_size;
|
||||
val.usec_timeout = usec_timeout;
|
||||
val.fb_bpp = fb_bpp;
|
||||
val.front_offset = front_offset;
|
||||
val.front_pitch = front_pitch;
|
||||
val.back_offset = back_offset;
|
||||
val.back_pitch = back_pitch;
|
||||
val.depth_bpp = depth_bpp;
|
||||
val.depth_offset = depth_offset;
|
||||
val.depth_pitch = depth_pitch;
|
||||
val.fb_offset = fb_offset;
|
||||
val.mmio_offset = mmio_offset;
|
||||
val.ring_offset = ring_offset;
|
||||
val.ring_rptr_offset = ring_rptr_offset;
|
||||
val.buffers_offset = buffers_offset;
|
||||
val.gart_textures_offset = gart_textures_offset;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_init_t(drm_radeon_init_t val) {
|
||||
sarea_priv_offset = val.sarea_priv_offset;
|
||||
is_pci = val.is_pci;
|
||||
cp_mode = val.cp_mode;
|
||||
gart_size = val.gart_size;
|
||||
ring_size = val.ring_size;
|
||||
usec_timeout = val.usec_timeout;
|
||||
fb_bpp = val.fb_bpp;
|
||||
front_offset = val.front_offset;
|
||||
front_pitch = val.front_pitch;
|
||||
back_offset = val.back_offset;
|
||||
back_pitch = val.back_pitch;
|
||||
depth_bpp = val.depth_bpp;
|
||||
depth_offset = val.depth_offset;
|
||||
depth_pitch = val.depth_pitch;
|
||||
fb_offset = val.fb_offset;
|
||||
mmio_offset = val.mmio_offset;
|
||||
ring_offset = val.ring_offset;
|
||||
ring_rptr_offset = val.ring_rptr_offset;
|
||||
buffers_offset = val.buffers_offset;
|
||||
gart_textures_offset = val.gart_textures_offset;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_clear")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_clear_t {
|
||||
uint32_t flags;
|
||||
uint32_t clear_color;
|
||||
uint32_t clear_depth;
|
||||
uint32_t color_mask;
|
||||
uint32_t depth_mask;
|
||||
compat_ptr<drm_radeon_clear_rect_t> depth_boxes;
|
||||
|
||||
fex_drm_radeon_clear_t() = delete;
|
||||
|
||||
operator drm_radeon_clear_t() const {
|
||||
drm_radeon_clear_t val{};
|
||||
val.flags = flags;
|
||||
val.clear_color = clear_color;
|
||||
val.clear_depth = clear_depth;
|
||||
val.color_mask = color_mask;
|
||||
val.depth_mask = depth_mask;
|
||||
val.depth_boxes = depth_boxes;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_clear_t(drm_radeon_clear_t val)
|
||||
: depth_boxes {val.depth_boxes} {
|
||||
flags = val.flags;
|
||||
clear_color = val.clear_color;
|
||||
clear_depth = val.clear_depth;
|
||||
color_mask = val.color_mask;
|
||||
depth_mask = val.depth_mask;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_stipple")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_stipple_t {
|
||||
compat_ptr<uint32_t> mask;
|
||||
|
||||
fex_drm_radeon_stipple_t() = delete;
|
||||
|
||||
operator drm_radeon_stipple_t() const {
|
||||
drm_radeon_stipple_t val{};
|
||||
val.mask = mask;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_stipple_t(drm_radeon_stipple_t val)
|
||||
: mask {val.mask} {
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_texture")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_texture_t {
|
||||
uint32_t offset;
|
||||
int32_t pitch;
|
||||
int32_t format;
|
||||
int32_t width;
|
||||
int32_t height;
|
||||
compat_ptr<drm_radeon_tex_image_t> image;
|
||||
|
||||
fex_drm_radeon_texture_t() = delete;
|
||||
|
||||
operator drm_radeon_texture_t() const {
|
||||
drm_radeon_texture_t val{};
|
||||
val.offset = offset;
|
||||
val.pitch = pitch;
|
||||
val.format = format;
|
||||
val.width = width;
|
||||
val.height = height;
|
||||
val.image = image;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_texture_t(drm_radeon_texture_t val)
|
||||
: image {val.image} {
|
||||
offset = val.offset;
|
||||
pitch = val.pitch;
|
||||
format = val.format;
|
||||
width = val.width;
|
||||
height = val.height;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_vertex2")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_vertex2_t {
|
||||
int32_t idx;
|
||||
int32_t discard;
|
||||
int32_t nr_states;
|
||||
compat_ptr<drm_radeon_state_t> state;
|
||||
int32_t nr_prims;
|
||||
compat_ptr<drm_radeon_prim_t> prim;
|
||||
|
||||
fex_drm_radeon_vertex2_t() = delete;
|
||||
|
||||
operator drm_radeon_vertex2_t() const {
|
||||
drm_radeon_vertex2_t val;
|
||||
val.idx = idx;
|
||||
val.discard = discard;
|
||||
val.nr_states = nr_states;
|
||||
val.state = state;
|
||||
val.nr_prims = nr_prims;
|
||||
val.prim = prim;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_vertex2_t(drm_radeon_vertex2_t val)
|
||||
: state {val.state}
|
||||
, prim {val.prim} {
|
||||
idx = val.idx;
|
||||
discard = val.discard;
|
||||
nr_states = val.nr_states;
|
||||
nr_prims = val.nr_prims;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_cmd_buffer")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_cmd_buffer_t {
|
||||
int32_t bufsz;
|
||||
compat_ptr<char> buf;
|
||||
int32_t nbox;
|
||||
compat_ptr<drm_clip_rect> boxes;
|
||||
|
||||
fex_drm_radeon_cmd_buffer_t() = delete;
|
||||
|
||||
operator drm_radeon_cmd_buffer_t() const {
|
||||
drm_radeon_cmd_buffer_t val;
|
||||
val.bufsz = bufsz;
|
||||
val.buf = buf;
|
||||
val.nbox = nbox;
|
||||
val.boxes = boxes;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_cmd_buffer_t(drm_radeon_cmd_buffer_t val)
|
||||
: buf {val.buf}
|
||||
, boxes {val.boxes} {
|
||||
val.bufsz = bufsz;
|
||||
val.nbox = nbox;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_getparam")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_getparam_t {
|
||||
int32_t param;
|
||||
compat_ptr<void> value;
|
||||
|
||||
fex_drm_radeon_getparam_t() = delete;
|
||||
|
||||
operator drm_radeon_getparam_t() const {
|
||||
drm_radeon_getparam_t val;
|
||||
val.param = param;
|
||||
val.value = value;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_getparam_t(drm_radeon_getparam_t val)
|
||||
: value {val.value} {
|
||||
val.param = param;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_mem_alloc")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_mem_alloc_t {
|
||||
int32_t region;
|
||||
int32_t alignment;
|
||||
int32_t size;
|
||||
compat_ptr<int32_t> region_offset;
|
||||
|
||||
fex_drm_radeon_mem_alloc_t() = delete;
|
||||
|
||||
operator drm_radeon_mem_alloc_t() const {
|
||||
drm_radeon_mem_alloc_t val;
|
||||
val.region = region;
|
||||
val.alignment = alignment;
|
||||
val.size = size;
|
||||
val.region_offset = region_offset;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_mem_alloc_t(drm_radeon_mem_alloc_t val)
|
||||
: region_offset {val.region_offset} {
|
||||
val.region = region;
|
||||
val.alignment = alignment;
|
||||
val.size = size;
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_irq_emit")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
fex_drm_radeon_irq_emit_t {
|
||||
compat_ptr<int32_t> irq_seq;
|
||||
|
||||
fex_drm_radeon_irq_emit_t() = delete;
|
||||
|
||||
operator drm_radeon_irq_emit_t() const {
|
||||
drm_radeon_irq_emit_t val;
|
||||
val.irq_seq = irq_seq;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_irq_emit_t(drm_radeon_irq_emit_t val)
|
||||
: irq_seq {val.irq_seq} {
|
||||
}
|
||||
};
|
||||
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_radeon_setparam")
|
||||
FEX_ANNOTATE("fex-match")
|
||||
FEX_PACKED
|
||||
fex_drm_radeon_setparam_t {
|
||||
uint32_t param;
|
||||
compat_int64_t value;
|
||||
|
||||
fex_drm_radeon_setparam_t() = delete;
|
||||
|
||||
operator drm_radeon_setparam_t() const {
|
||||
drm_radeon_setparam_t val;
|
||||
val.param = param;
|
||||
val.value = value;
|
||||
return val;
|
||||
}
|
||||
|
||||
fex_drm_radeon_setparam_t(drm_radeon_setparam_t val) {
|
||||
param = val.param;
|
||||
value = val.value;
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
namespace MSM {
|
||||
struct
|
||||
FEX_ANNOTATE("alias-x86_32-drm_msm_timespec")
|
||||
@@ -1061,6 +1416,7 @@ fex_drm_v3d_submit_csd {
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/lima_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/panfrost_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/nouveau_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/radeon_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/vc4_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/v3d_drm.inl"
|
||||
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_CP_INIT, DRM_IOW(DRM_COMMAND_BASE + DRM_RADEON_CP_INIT, FEX::HLE::x32::RADEON::fex_drm_radeon_init_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_CP_START)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_CP_STOP)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_CP_RESET)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_CP_IDLE)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_RESET)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_FULLSCREEN)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_SWAP)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_CLEAR, DRM_IOW(DRM_COMMAND_BASE + DRM_RADEON_CLEAR, FEX::HLE::x32::RADEON::fex_drm_radeon_clear_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_VERTEX)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_INDICES)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_STIPPLE, DRM_IOW( DRM_COMMAND_BASE + DRM_RADEON_STIPPLE, FEX::HLE::x32::RADEON::fex_drm_radeon_stipple_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_INDIRECT)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_TEXTURE, DRM_IOWR(DRM_COMMAND_BASE + DRM_RADEON_TEXTURE, FEX::HLE::x32::RADEON::fex_drm_radeon_texture_t))
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_VERTEX2, DRM_IOW(DRM_COMMAND_BASE + DRM_RADEON_VERTEX2, FEX::HLE::x32::RADEON::fex_drm_radeon_vertex2_t))
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_CMDBUF, DRM_IOW(DRM_COMMAND_BASE + DRM_RADEON_CMDBUF, FEX::HLE::x32::RADEON::fex_drm_radeon_cmd_buffer_t))
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_GETPARAM, DRM_IOWR(DRM_COMMAND_BASE + DRM_RADEON_GETPARAM, FEX::HLE::x32::RADEON::fex_drm_radeon_getparam_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_FLIP)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_ALLOC, DRM_IOWR(DRM_COMMAND_BASE + DRM_RADEON_ALLOC, FEX::HLE::x32::RADEON::fex_drm_radeon_mem_alloc_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_FREE)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_INIT_HEAP)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_IRQ_EMIT, DRM_IOWR(DRM_COMMAND_BASE + DRM_RADEON_IRQ_EMIT, FEX::HLE::x32::RADEON::fex_drm_radeon_irq_emit_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_IRQ_WAIT)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_CP_RESUME)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_SETPARAM, DRM_IOW(DRM_COMMAND_BASE + DRM_RADEON_SETPARAM, FEX::HLE::x32::RADEON::fex_drm_radeon_setparam_t))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_SURF_ALLOC)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_SURF_FREE)
|
||||
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_INFO)
|
||||
_CUSTOM_META(DRM_IOCTL_RADEON_GEM_CREATE, DRM_IOWR(DRM_COMMAND_BASE + DRM_RADEON_GEM_CREATE, FEX::HLE::x32::RADEON::fex_drm_radeon_gem_create))
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_MMAP)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_PREAD)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_PWRITE)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_SET_DOMAIN)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_WAIT_IDLE)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_CS)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_INFO)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_SET_TILING)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_GET_TILING)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_BUSY)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_VA)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_OP)
|
||||
_BASIC_META(DRM_IOCTL_RADEON_GEM_USERPTR)
|
||||
|
||||
@@ -178,6 +178,141 @@ namespace FEX::HLE::x32 {
|
||||
return -EPERM;
|
||||
}
|
||||
|
||||
uint32_t RADEON_Handler(int fd, uint32_t cmd, uint32_t args) {
|
||||
switch (_IOC_NR(cmd)) {
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_CP_INIT): {
|
||||
RADEON::fex_drm_radeon_init_t *val = reinterpret_cast<RADEON::fex_drm_radeon_init_t*>(args);
|
||||
drm_radeon_init_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_CP_INIT, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_CLEAR): {
|
||||
RADEON::fex_drm_radeon_clear_t *val = reinterpret_cast<RADEON::fex_drm_radeon_clear_t*>(args);
|
||||
drm_radeon_clear_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_CLEAR, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_STIPPLE): {
|
||||
RADEON::fex_drm_radeon_stipple_t *val = reinterpret_cast<RADEON::fex_drm_radeon_stipple_t*>(args);
|
||||
drm_radeon_stipple_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_STIPPLE, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_TEXTURE): {
|
||||
RADEON::fex_drm_radeon_texture_t *val = reinterpret_cast<RADEON::fex_drm_radeon_texture_t*>(args);
|
||||
drm_radeon_texture_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_TEXTURE, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_VERTEX2): {
|
||||
RADEON::fex_drm_radeon_vertex2_t *val = reinterpret_cast<RADEON::fex_drm_radeon_vertex2_t*>(args);
|
||||
drm_radeon_vertex2_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_VERTEX2, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_CMDBUF): {
|
||||
RADEON::fex_drm_radeon_cmd_buffer_t *val = reinterpret_cast<RADEON::fex_drm_radeon_cmd_buffer_t*>(args);
|
||||
drm_radeon_cmd_buffer_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_CMDBUF, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_GETPARAM): {
|
||||
RADEON::fex_drm_radeon_getparam_t *val = reinterpret_cast<RADEON::fex_drm_radeon_getparam_t*>(args);
|
||||
drm_radeon_getparam_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_GETPARAM, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_ALLOC): {
|
||||
RADEON::fex_drm_radeon_mem_alloc_t *val = reinterpret_cast<RADEON::fex_drm_radeon_mem_alloc_t*>(args);
|
||||
drm_radeon_mem_alloc_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_ALLOC, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_IRQ_EMIT): {
|
||||
RADEON::fex_drm_radeon_irq_emit_t *val = reinterpret_cast<RADEON::fex_drm_radeon_irq_emit_t*>(args);
|
||||
drm_radeon_irq_emit_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_IRQ_EMIT, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_SETPARAM): {
|
||||
RADEON::fex_drm_radeon_setparam_t *val = reinterpret_cast<RADEON::fex_drm_radeon_setparam_t*>(args);
|
||||
drm_radeon_setparam_t Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_SETPARAM, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
case _IOC_NR(FEX_DRM_IOCTL_RADEON_GEM_CREATE): {
|
||||
RADEON::fex_drm_radeon_gem_create *val = reinterpret_cast<RADEON::fex_drm_radeon_gem_create*>(args);
|
||||
drm_radeon_gem_create Host_val = *val;
|
||||
uint64_t Result = ioctl(fd, DRM_IOCTL_RADEON_GEM_CREATE, &Host_val);
|
||||
if (Result != -1) {
|
||||
*val = Host_val;
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
#define _BASIC_META(x) case _IOC_NR(x):
|
||||
#define _BASIC_META_VAR(x, args...) case _IOC_NR(x):
|
||||
#define _CUSTOM_META(name, ioctl_num)
|
||||
#define _CUSTOM_META_OFFSET(name, ioctl_num, offset)
|
||||
// DRM
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/radeon_drm.inl"
|
||||
{
|
||||
uint64_t Result = ::ioctl(fd, cmd, args);
|
||||
SYSCALL_ERRNO();
|
||||
break;
|
||||
}
|
||||
default:
|
||||
UnhandledIoctl("RADEON", fd, cmd, args);
|
||||
return -EPERM;
|
||||
break;
|
||||
}
|
||||
#undef _BASIC_META
|
||||
#undef _BASIC_META_VAR
|
||||
#undef _CUSTOM_META
|
||||
#undef _CUSTOM_META_OFFSET
|
||||
return -EPERM;
|
||||
}
|
||||
|
||||
uint32_t MSM_Handler(int fd, uint32_t cmd, uint32_t args) {
|
||||
switch (_IOC_NR(cmd)) {
|
||||
case _IOC_NR(FEX_DRM_IOCTL_MSM_WAIT_FENCE): {
|
||||
@@ -435,6 +570,9 @@ namespace FEX::HLE::x32 {
|
||||
if (strcmp(Version.name, "amdgpu") == 0) {
|
||||
FDToHandler.SetFDHandler(fd, AMDGPU_Handler);
|
||||
}
|
||||
else if (strcmp(Version.name, "radeon") == 0) {
|
||||
FDToHandler.SetFDHandler(fd, RADEON_Handler);
|
||||
}
|
||||
else if (strcmp(Version.name, "msm") == 0) {
|
||||
FDToHandler.SetFDHandler(fd, MSM_Handler);
|
||||
}
|
||||
@@ -611,6 +749,7 @@ namespace FEX::HLE::x32 {
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/lima_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/panfrost_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/nouveau_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/radeon_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/vc4_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/v3d_drm.inl"
|
||||
#include "Tests/LinuxSyscalls/x32/Ioctl/virtio_drm.inl"
|
||||
|
||||
Loaded 100 of 207 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user