mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-10 12:00:18 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8b1383d235 | ||
|
|
a73fab3bb5 | ||
|
|
7bd64d9c53 | ||
|
|
1786c2f157 | ||
|
|
958b671736 | ||
|
|
6549b66cf6 | ||
|
|
0c855a5ce3 | ||
|
|
6864d48dcf | ||
|
|
99816a23a8 | ||
|
|
2ff9546523 | ||
|
|
06541f21d6 | ||
|
|
45a37edd4a | ||
|
|
2a713a1f51 | ||
|
|
3e104de377 | ||
|
|
b22b316e70 | ||
|
|
df461546c5 | ||
|
|
1c1c43cd86 | ||
|
|
3eac9f937e | ||
|
|
f13a0d8e84 | ||
|
|
ca12dc9213 | ||
|
|
365ed2cd70 | ||
|
|
7efbfed0bf | ||
|
|
ed502738c6 | ||
|
|
ba162bb058 | ||
|
|
f194d35913 | ||
|
|
e81c84e00a | ||
|
|
68dc9030bc | ||
|
|
fedad275e7 | ||
|
|
f6b4c76d76 | ||
|
|
27854aa091 | ||
|
|
d966ae145e | ||
|
|
4cb37e6a1b | ||
|
|
d9da81e99b | ||
|
|
7d734740be | ||
|
|
06299cca4b | ||
|
|
af5aaab38b | ||
|
|
77bb01d384 | ||
|
|
a9eb1bd4e6 | ||
|
|
83ef2da95f | ||
|
|
ba13ccadb8 | ||
|
|
4a2dee873d | ||
|
|
589906e6e6 | ||
|
|
f0fcf6d9e6 | ||
|
|
1591ced5a7 | ||
|
|
0d235d63d0 | ||
|
|
820d5c1447 | ||
|
|
8093e8f5f1 | ||
|
|
802857e411 | ||
|
|
bf3a1839a2 | ||
|
|
ad7844d7da | ||
|
|
3ff9128a8a | ||
|
|
c865eb98ea | ||
|
|
217a9228f8 | ||
|
|
105d0a36ad | ||
|
|
263279d5dd | ||
|
|
861ecbec0f | ||
|
|
5ff9bb3669 | ||
|
|
e794584bb5 | ||
|
|
a792dd0703 | ||
|
|
a9a6a645bf | ||
|
|
7c93becd5f | ||
|
|
4bbaef58e9 | ||
|
|
95791a985a | ||
|
|
0dfefe9730 | ||
|
|
8481c797df | ||
|
|
a503b5e20b | ||
|
|
4078840ef1 | ||
|
|
ab51958b26 | ||
|
|
3376587b6a | ||
|
|
6681d7dcf9 | ||
|
|
34224481c6 | ||
|
|
1837aaabe4 | ||
|
|
5811914a78 | ||
|
|
a109a4efa1 | ||
|
|
a08a6ce5de | ||
|
|
3dc8a3ddc1 | ||
|
|
5e103365f7 | ||
|
|
cdea8d7f74 | ||
|
|
ef6dc3d802 | ||
|
|
e6edb349ba | ||
|
|
ad132267ec | ||
|
|
6f837281ef | ||
|
|
2a1d29d2df | ||
|
|
bdceb4ca89 | ||
|
|
656bb928cf | ||
|
|
0fbe69ebcf | ||
|
|
ece817c691 | ||
|
|
3fbc8204b7 | ||
|
|
44a5481254 | ||
|
|
cf2ff90f87 | ||
|
|
b8dd5d95b0 | ||
|
|
7cd52febc2 | ||
|
|
8e079c1965 | ||
|
|
4928af5a64 | ||
|
|
289df740cd | ||
|
|
6ad7392cd7 | ||
|
|
b15d5f299c | ||
|
|
5c7c959dd9 | ||
|
|
6927c7577a | ||
|
|
88682a457a | ||
|
|
61ae53cc03 | ||
|
|
0e67f30103 | ||
|
|
babd6e9a7b | ||
|
|
76caa2c6e3 | ||
|
|
dc9f8aa855 | ||
|
|
ded8b3284a | ||
|
|
2e24ee7a5f | ||
|
|
048ae597d2 | ||
|
|
0147f7aa19 | ||
|
|
23cda2c961 | ||
|
|
feb67658e1 | ||
|
|
65ee1fafa8 | ||
|
|
49f8332c5b | ||
|
|
afce108ed7 | ||
|
|
30f3b545af | ||
|
|
bf1597920c | ||
|
|
f66bf3811e | ||
|
|
a01e29ac99 | ||
|
|
3136a5e2f8 | ||
|
|
c4f00a05df | ||
|
|
e9a8f8a9ff | ||
|
|
261ae7a195 | ||
|
|
7eaf5ae9e0 | ||
|
|
10a02449b1 | ||
|
|
debc57e8c7 | ||
|
|
7a4fff8e5b | ||
|
|
ec0a2a8671 | ||
|
|
b8b516f7b6 | ||
|
|
3b1e91b1fc | ||
|
|
8532593d91 | ||
|
|
55bd16e2c8 | ||
|
|
c7797d56c9 | ||
|
|
9b573effd1 | ||
|
|
dfd1aedae5 | ||
|
|
221ae2d7b4 | ||
|
|
9127d206b5 | ||
|
|
ecc6fea54e | ||
|
|
99446da7c1 | ||
|
|
ad75563a26 | ||
|
|
197af972d9 | ||
|
|
fbd706c191 | ||
|
|
6b85fa5611 | ||
|
|
fae66a921f | ||
|
|
e6fc462e9d | ||
|
|
d11a265a53 | ||
|
|
395d870814 | ||
|
|
fec1ffaa6b | ||
|
|
d4fdb28e72 | ||
|
|
39a5c2021e | ||
|
|
f41501444d | ||
|
|
86b26b80ce | ||
|
|
ae9a5b1125 | ||
|
|
38579807c2 | ||
|
|
c19119bcd6 | ||
|
|
47f1ad693d | ||
|
|
2164d7bb96 | ||
|
|
c326e2d669 | ||
|
|
8eaf45414c | ||
|
|
b3297d106e | ||
|
|
2d0e19e6a7 | ||
|
|
fd2ee4dc46 | ||
|
|
b1bbc37c59 | ||
|
|
56409d4f2b | ||
|
|
ec0683b729 | ||
|
|
89e5041e70 | ||
|
|
8b3e7312e0 | ||
|
|
3c76a9176d | ||
|
|
a37def2c22 |
No files matched your search
+5
-2
@@ -38,7 +38,10 @@ option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
if (NOT DATA_DIRECTORY)
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
|
||||
endif()
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
@@ -414,7 +417,7 @@ if (TUNE_CPU STREQUAL "native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
elseif (NOT TUNE_CPU STREQUAL "none")
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
|
||||
|
||||
@@ -3952,7 +3952,7 @@ public:
|
||||
L = (Index >> 0) & 1;
|
||||
M = 0;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T>, "Can't encode DRegister with i64Bit");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T>), "Can't encode DRegister with i64Bit");
|
||||
// Index encoded in H
|
||||
H = Index;
|
||||
L = 0;
|
||||
@@ -3983,7 +3983,7 @@ public:
|
||||
L = (Index >> 0) & 1;
|
||||
M = 0;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T>, "Can't encode DRegister with i64Bit");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T>), "Can't encode DRegister with i64Bit");
|
||||
// Index encoded in H
|
||||
H = Index;
|
||||
L = 0;
|
||||
@@ -4014,7 +4014,7 @@ public:
|
||||
L = (Index >> 0) & 1;
|
||||
M = 0;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T>, "Can't encode DRegister with i64Bit");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T>), "Can't encode DRegister with i64Bit");
|
||||
// Index encoded in H
|
||||
H = Index;
|
||||
L = 0;
|
||||
|
||||
@@ -1648,8 +1648,8 @@ public:
|
||||
template<typename T>
|
||||
void ASIMDLoadStoreSinglePost(uint32_t Op, uint32_t Q, uint32_t L, uint32_t R, uint32_t opcode, uint32_t S, uint32_t size,
|
||||
ARMEmitter::Register rm, ARMEmitter::Register rn, T rt) {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T> || std::is_same_v<ARMEmitter::DRegister, T>, "Only supports 128-bit and "
|
||||
"64-bit vector registers.");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T> || std::is_same_v<ARMEmitter::DRegister, T>), "Only supports 128-bit and "
|
||||
"64-bit vector registers.");
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Q << 30;
|
||||
|
||||
@@ -575,7 +575,7 @@ def print_ir_arg_printer():
|
||||
|
||||
if arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}], RAData);\n".format(SSAArgNum))
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}]);\n".format(SSAArgNum))
|
||||
SSAArgNum = SSAArgNum + 1
|
||||
else:
|
||||
# User defined op that is stored
|
||||
@@ -587,6 +587,15 @@ def print_ir_arg_printer():
|
||||
output_file.write("#undef IROP_ARGPRINTER_HELPER\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
def print_validation(op):
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
# Print out IR allocator helpers
|
||||
def print_ir_allocator_helpers():
|
||||
output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n")
|
||||
@@ -678,7 +687,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
@@ -708,35 +717,16 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\tauto _Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t_Op.first->{} = {}->Wrapped(ListDataBegin);\n".format(arg.Name, arg.Name))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
output_file.write("\t\t_Op.first->{} = {};\n".format(arg.Name, arg.Name))
|
||||
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if not arg.Temporary and not arg.IsSSA:
|
||||
output_file.write("\t\t_Op.first->{} = {};\n".format(arg.Name, arg.Name))
|
||||
|
||||
if (op.HasDest):
|
||||
# We can only infer a size if we have arguments
|
||||
if op.DestSize == None:
|
||||
# We need to infer destination size
|
||||
output_file.write("\t\tIR::OpSize InferSize = OpSize::iUnsized;\n")
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tauto Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tInferSize = std::max(InferSize, Size{});\n".format(arg.Name))
|
||||
|
||||
output_file.write("\t\t_Op.first->Header.Size = InferSize;\n")
|
||||
assert not (op.HasDest and op.DestSize is None)
|
||||
|
||||
# Some ops without a destination still need an operating size
|
||||
# Effectively reusing the destination size value for operation size
|
||||
@@ -748,18 +738,64 @@ def print_ir_allocator_helpers():
|
||||
else:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = {};\n".format(op.ElementSize))
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
# Only validate here if there's no OrderedNode * version. Else
|
||||
# validation is in that version, see the comment below.
|
||||
if op.SSAArgNum == 0:
|
||||
print_validation(op)
|
||||
|
||||
output_file.write("\t\treturn _Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
# Now do the OrderedNode * version if necessary
|
||||
if op.SSAArgNum:
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
|
||||
output_file.write(") {\n")
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
|
||||
# Insert validation here. This is skipped for the
|
||||
# OrderedNodeWrapper version because validation can depend on
|
||||
# the OrderedNode, but that's ok in practice. Everything pre-RA
|
||||
# uses the OrderedNode version, and anything RA-onwards is
|
||||
# dubious to validate.
|
||||
print_validation(op)
|
||||
|
||||
output_file.write(f"\t\treturn _{op.Name}(")
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
output_file.write(arg.Name)
|
||||
if arg.IsSSA:
|
||||
output_file.write("->Wrapped(ListDataBegin)")
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
output_file.write(");\n");
|
||||
output_file.write("\t}\n\n");
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
|
||||
@@ -69,7 +69,6 @@ set (SRCS
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
|
||||
@@ -118,7 +118,7 @@
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks/",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks/",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
@@ -134,7 +134,7 @@
|
||||
},
|
||||
"ThunkHostLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks_32/",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit host-side thunking libraries."
|
||||
]
|
||||
|
||||
@@ -50,7 +50,6 @@ namespace HLE {
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
struct IRListCopy;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
@@ -76,7 +75,7 @@ struct CustomIRResult {
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
@@ -165,7 +164,8 @@ public:
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
@@ -264,6 +264,10 @@ public:
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// NOTE: Other threads sharing the same CodeBuffer may reference
|
||||
// invalidated data ranges through their L1/L2 caches. This is
|
||||
// not currently a problem since FEX does not repurpose the
|
||||
// invalidated CodeBuffer memory range currently.
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
@@ -271,7 +275,6 @@ public:
|
||||
|
||||
struct GenerateIRResult {
|
||||
std::optional<IR::IRListView> IRView;
|
||||
IR::RegisterAllocationData* RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
@@ -299,6 +302,7 @@ public:
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
|
||||
@@ -34,7 +34,15 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
uint32_t Bits = IR::OpSizeAsBits(A.AddrSize);
|
||||
|
||||
if (A.Base || A.Index) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, Bits, 0, Tmp);
|
||||
} else if (A.Offset) {
|
||||
uint64_t X = A.Offset;
|
||||
X &= (1ull << Bits) - 1;
|
||||
Tmp = IREmit->_Constant(X);
|
||||
}
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
@@ -155,4 +163,4 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
}
|
||||
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -501,7 +501,10 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && ARMEmitter::Emitter::IsInt32(AlignedOffset)) {
|
||||
// NOTE: JIT output is moved to a different buffer after compilation, so the
|
||||
// current cursor address doesn't match the runtime instruction address.
|
||||
// Hence this optimization is disabled until we enable code relocation patches.
|
||||
if (RequiredMoveSegments > 1 && ARMEmitter::Emitter::IsInt32(AlignedOffset) && false) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
@@ -590,8 +593,8 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
|
||||
void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
constexpr static std::array< std::tuple<ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
}};
|
||||
|
||||
for (auto& RegQuad : FPRs) {
|
||||
|
||||
@@ -6,6 +6,8 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
@@ -13,6 +15,10 @@
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
@@ -264,10 +270,9 @@ namespace CPU {
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
CPUBackend::CPUBackend(CodeBufferManager& CodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
|
||||
: ThreadState(ThreadState)
|
||||
, InitialCodeSize(InitialCodeSize)
|
||||
, MaxCodeSize(MaxCodeSize) {
|
||||
, CodeBuffers(CodeBuffers) {
|
||||
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
@@ -304,52 +309,63 @@ namespace CPU {
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
CPUBackend::~CPUBackend() = default;
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = CodeBuffers.data();
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
RegisterForSignalHandler(PrevCodeBuffer);
|
||||
return CurrentCodeBuffer.get();
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
void CPUBackend::RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer> CodeBuffer) {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter != 0) {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Keep a reference to the old code buffer to delay deallocation
|
||||
SignalHandlerCodeBuffers.push_back(CodeBuffer);
|
||||
} else {
|
||||
SignalHandlerCodeBuffers.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
|
||||
auto NewCodeBuffer = CodeBuffers.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
|
||||
return *Buffer.LookupCache;
|
||||
}
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size)
|
||||
: Size(Size) {
|
||||
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, Size);
|
||||
}
|
||||
|
||||
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
@@ -375,39 +391,51 @@ namespace CPU {
|
||||
}
|
||||
#endif
|
||||
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
OnCodeBufferAllocated(*Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
|
||||
if (!Latest) {
|
||||
AllocateNew(INITIAL_CODE_SIZE);
|
||||
}
|
||||
return Latest;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer() {
|
||||
if (!Latest) {
|
||||
// Allocate initial CodeBuffer and return it
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->Size;
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
for (auto& Buffer : CodeBuffers) {
|
||||
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr) {
|
||||
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
|
||||
};
|
||||
|
||||
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
for (auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
if (CheckCodeBuffer(*Buffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -9,6 +9,8 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
@@ -18,7 +20,6 @@ namespace FEXCore {
|
||||
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
} // namespace IR
|
||||
|
||||
namespace Core {
|
||||
@@ -32,19 +33,61 @@ namespace CodeSerialize {
|
||||
struct CodeObjectFileSection;
|
||||
}
|
||||
|
||||
struct GuestToHostMap;
|
||||
|
||||
namespace CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
CodeBuffer(CodeBuffer&& oth) = delete;
|
||||
CodeBuffer& operator=(CodeBuffer&&) = delete;
|
||||
|
||||
~CodeBuffer();
|
||||
};
|
||||
|
||||
/**
|
||||
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
|
||||
*
|
||||
* The CodeBuffer is managed as a partially persistent data structure:
|
||||
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
|
||||
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
|
||||
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
|
||||
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
|
||||
*/
|
||||
class CodeBufferManager {
|
||||
public:
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
|
||||
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
|
||||
|
||||
// Write offset into the latest CodeBuffer
|
||||
std::size_t LatestOffset;
|
||||
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
/**
|
||||
* @param InitialCodeSize - Initial size for the code buffers
|
||||
* @param MaxCodeSize - Max size for the code buffers
|
||||
*/
|
||||
CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
|
||||
@@ -119,7 +162,7 @@ namespace CPU {
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) = 0;
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
@@ -143,6 +186,11 @@ namespace CPU {
|
||||
|
||||
bool IsAddressInCodeBuffer(uintptr_t Address) const;
|
||||
|
||||
// Updates the CodeBuffer if needed and returns a reference to the old one.
|
||||
// The returned reference should be kept alive carefully to avoid early deletion of resources.
|
||||
[[nodiscard]]
|
||||
fextl::shared_ptr<CodeBuffer> CheckCodeBufferUpdate();
|
||||
|
||||
protected:
|
||||
// Max spill slot size in bytes. We need at most 32 bytes
|
||||
// to be able to handle a 256-bit vector store to a slot.
|
||||
@@ -150,24 +198,21 @@ namespace CPU {
|
||||
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
size_t InitialCodeSize, MaxCodeSize;
|
||||
size_t MaxCodeSize;
|
||||
[[nodiscard]]
|
||||
CodeBuffer* GetEmptyCodeBuffer();
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer* CurrentCodeBuffer {};
|
||||
// This is the code buffer containing the main code under execution by this thread.
|
||||
// CheckCodeBufferUpdate must be used before compiling new code.
|
||||
fextl::shared_ptr<CodeBuffer> CurrentCodeBuffer;
|
||||
|
||||
// Old CodeBuffer generations required to be valid until returning from signal handlers
|
||||
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
|
||||
|
||||
CodeBufferManager& CodeBuffers;
|
||||
|
||||
private:
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
// This is the array of code buffers. Unless signals force us to keep more than
|
||||
// buffer, there will be only one entry here
|
||||
fextl::vector<CodeBuffer> CodeBuffers {};
|
||||
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
|
||||
};
|
||||
|
||||
} // namespace CPU
|
||||
|
||||
@@ -115,7 +115,7 @@ public:
|
||||
|
||||
private:
|
||||
const FEXCore::Context::ContextImpl* CTX;
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
[[maybe_unused]] bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
@@ -495,7 +495,13 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
@@ -503,18 +509,22 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LookupCache->ClearCache();
|
||||
Thread->CPUBackend->ClearCache();
|
||||
if (NewCodeBuffer) {
|
||||
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
|
||||
FEXCore::File::File FD = FEXCore::File::File::GetStdERR();
|
||||
fextl::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
FEXCore::IR::Dump(&out, &NewIR);
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
@@ -686,7 +696,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {{}, nullptr, 0, 0, 0, 0};
|
||||
return {{}, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -712,22 +722,19 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto ShouldDump = Thread->OpDispatcher->ShouldDumpIR();
|
||||
// Debug
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, nullptr);
|
||||
IRDumper(Thread, IREmitter, GuestRIP);
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IREmitter);
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr;
|
||||
|
||||
// Debug
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, RAData);
|
||||
IRDumper(Thread, IREmitter, GuestRIP);
|
||||
}
|
||||
|
||||
return {
|
||||
.IRView = IREmitter->ViewIR(),
|
||||
.RAData = RAData,
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
@@ -760,8 +767,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
|
||||
// Generate IR + Meta Info
|
||||
auto [IRView, RAData, TotalInstructions, TotalInstructionsLength, StartAddr, Length] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
return {nullptr, nullptr, 0, 0};
|
||||
}
|
||||
@@ -772,7 +778,19 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), RAData, TFSet);
|
||||
|
||||
// Re-check if another thread raced us in compiling this block.
|
||||
// We could lock CodeBufferWriteMutex earlier to prevent this from happening,
|
||||
// but this would increase lock contention. Redundant frontend runs aren't
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {.CompiledCode = reinterpret_cast<uint8_t*>(Block), .DebugData = nullptr, .StartAddr = 0, .Length = 0};
|
||||
}
|
||||
}
|
||||
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), TFSet);
|
||||
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
@@ -807,6 +825,9 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
} else if (!DebugData) {
|
||||
// DebugData wasn't populated, indicating another thread raced us for compiling this block
|
||||
return reinterpret_cast<uintptr_t>(CodePtr);
|
||||
}
|
||||
|
||||
// The core managed to compile the code.
|
||||
@@ -888,7 +909,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
|
||||
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
|
||||
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
@@ -915,8 +936,8 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
|
||||
std::lock_guard<std::recursive_mutex> lkLookupCache(Thread->LookupCache->WriteLock);
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
|
||||
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
@@ -959,7 +980,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
auto Result = AddCustomIREntrypoint(
|
||||
Entrypoint,
|
||||
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
@@ -972,7 +993,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->_Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
ThunkHandler, (void*)GuestThunkEntrypoint);
|
||||
|
||||
|
||||
@@ -87,7 +87,7 @@ private:
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
const uint8_t* InstStream {};
|
||||
|
||||
@@ -16,18 +16,16 @@ namespace FEXCore::CPU {
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
\
|
||||
uint64_t Const; \
|
||||
if (IsInlineConstant(Op->Src2, &Const)) { \
|
||||
ConstOp(ConvertSize(IROp), GetReg(Node), GetReg(Op->Src1.ID()), Const); \
|
||||
} else { \
|
||||
VarOp(ConvertSize(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID())); \
|
||||
} \
|
||||
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
\
|
||||
uint64_t Const; \
|
||||
if (IsInlineConstant(Op->Src2, &Const)) { \
|
||||
ConstOp(ConvertSize(IROp), GetReg(Node), GetReg(Op->Src1), Const); \
|
||||
} else { \
|
||||
VarOp(ConvertSize(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2)); \
|
||||
} \
|
||||
}
|
||||
|
||||
DEF_BINOP_WITH_CONSTANT(Add, add, add)
|
||||
@@ -88,14 +86,14 @@ DEF_OP(CycleCounter) {
|
||||
DEF_OP(AddShift) {
|
||||
auto Op = IROp->C<IR::IROp_AddShift>();
|
||||
|
||||
add(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
add(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(AddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AddNZCV>();
|
||||
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -105,22 +103,22 @@ DEF_OP(AddNZCV) {
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmn(EmitSize, Src1, GetReg(Op->Src2.ID()));
|
||||
cmn(EmitSize, Src1, GetReg(Op->Src2));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AdcNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AdcNZCV>();
|
||||
|
||||
adcs(ConvertSize48(IROp), ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
adcs(ConvertSize48(IROp), ARMEmitter::Reg::zr, GetReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(AdcWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AdcWithFlags>();
|
||||
|
||||
adcs(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
adcs(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(AdcZeroWithFlags) {
|
||||
@@ -128,38 +126,38 @@ DEF_OP(AdcZeroWithFlags) {
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cset(Size, TMP1, ARMEmitter::Condition::CC_CC);
|
||||
adds(Size, GetReg(Node), GetReg(Op->Src1.ID()), TMP1);
|
||||
adds(Size, GetReg(Node), GetReg(Op->Src1), TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(AdcZero) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZero>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cinc(Size, GetReg(Node), GetReg(Op->Src1.ID()), ARMEmitter::Condition::CC_CC);
|
||||
cinc(Size, GetReg(Node), GetReg(Op->Src1), ARMEmitter::Condition::CC_CC);
|
||||
}
|
||||
|
||||
DEF_OP(Adc) {
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
|
||||
adc(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
adc(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(SbbWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_SbbWithFlags>();
|
||||
|
||||
sbcs(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
sbcs(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(SbbNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SbbNZCV>();
|
||||
|
||||
sbcs(ConvertSize48(IROp), ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
sbcs(ConvertSize48(IROp), ARMEmitter::Reg::zr, GetReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(Sbb) {
|
||||
auto Op = IROp->C<IR::IROp_Sbb>();
|
||||
|
||||
sbc(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
sbc(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
@@ -167,7 +165,7 @@ DEF_OP(TestNZ) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
@@ -179,7 +177,7 @@ DEF_OP(TestNZ) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, TMP1, Src1, Const);
|
||||
} else {
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
auto Src2 = GetReg(Op->Src2);
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
}
|
||||
|
||||
@@ -192,7 +190,7 @@ DEF_OP(TestNZ) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
tst(EmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
@@ -205,14 +203,14 @@ DEF_OP(TestZ) {
|
||||
|
||||
uint64_t Const;
|
||||
uint64_t Mask = IROp->Size == IR::OpSize::i64Bit ? ~0ULL : ((1ull << IR::OpSizeAsBits(IROp->Size)) - 1);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
|
||||
LOGMAN_THROW_A_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
if (Src1 == Src2) {
|
||||
tst(EmitSize, Src1 /* Src2 */, Mask);
|
||||
} else {
|
||||
@@ -225,7 +223,7 @@ DEF_OP(TestZ) {
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
sub(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
sub(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
@@ -236,7 +234,7 @@ DEF_OP(SubNZCV) {
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_A_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
cmp(EmitSize, GetReg(Op->Src1), Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < IR::OpSize::i32Bit ? (32 - IR::OpSizeAsBits(OpSize)) : 0;
|
||||
ARMEmitter::Register ShiftedSrc1 = GetZeroableReg(Op->Src1);
|
||||
@@ -249,9 +247,9 @@ DEF_OP(SubNZCV) {
|
||||
}
|
||||
|
||||
if (OpSize < IR::OpSize::i32Bit) {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()));
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -264,8 +262,8 @@ DEF_OP(CmpPairZ) {
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Compare, setting Z and clobbering NzCV
|
||||
cmp(EmitSize, GetReg(Op->Src1Lo.ID()), GetReg(Op->Src2Lo.ID()));
|
||||
ccmp(EmitSize, GetReg(Op->Src1Hi.ID()), GetReg(Op->Src2Hi.ID()), ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, GetReg(Op->Src1Lo), GetReg(Op->Src2Lo));
|
||||
ccmp(EmitSize, GetReg(Op->Src1Hi), GetReg(Op->Src2Hi), ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
// Restore NzCV
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
@@ -297,9 +295,9 @@ DEF_OP(SetSmallNZV) {
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
setf8(GetReg(Op->Src.ID()).W());
|
||||
setf8(GetReg(Op->Src).W());
|
||||
} else {
|
||||
setf16(GetReg(Op->Src.ID()).W());
|
||||
setf16(GetReg(Op->Src).W());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -316,7 +314,7 @@ DEF_OP(AXFlag) {
|
||||
//
|
||||
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
|
||||
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
|
||||
auto V_inv = GetReg(IROp->Args[0].ID());
|
||||
auto V_inv = GetReg(IROp->Args[0]);
|
||||
csetm(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Condition::CC_EQ);
|
||||
ccmn(ARMEmitter::Size::i64Bit, V_inv, TMP1, ARMEmitter::StatusFlags {0x2} /* nzCv */, ARMEmitter::Condition::CC_LE);
|
||||
}
|
||||
@@ -324,7 +322,7 @@ DEF_OP(AXFlag) {
|
||||
|
||||
DEF_OP(Parity) {
|
||||
auto Op = IROp->C<IR::IROp_Parity>();
|
||||
auto Raw = GetReg(Op->Raw.ID());
|
||||
auto Raw = GetReg(Op->Raw);
|
||||
auto Dest = GetReg(Node);
|
||||
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
@@ -353,7 +351,7 @@ DEF_OP(CondAddNZCV) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmn(ConvertSize48(IROp), Src1, Const, Flags, MapCC(Op->Cond));
|
||||
} else {
|
||||
ccmn(ConvertSize48(IROp), Src1, GetReg(Op->Src2.ID()), Flags, MapCC(Op->Cond));
|
||||
ccmn(ConvertSize48(IROp), Src1, GetReg(Op->Src2), Flags, MapCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -367,7 +365,7 @@ DEF_OP(CondSubNZCV) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmp(ConvertSize48(IROp), Src1, Const, Flags, MapCC(Op->Cond));
|
||||
} else {
|
||||
ccmp(ConvertSize48(IROp), Src1, GetReg(Op->Src2.ID()), Flags, MapCC(Op->Cond));
|
||||
ccmp(ConvertSize48(IROp), Src1, GetReg(Op->Src2), Flags, MapCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -375,32 +373,32 @@ DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
|
||||
if (Op->Cond == FEXCore::IR::COND_AL) {
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src.ID()));
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
|
||||
} else {
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src.ID()), MapCC(Op->Cond));
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
|
||||
mul(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
mul(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(UMul) {
|
||||
auto Op = IROp->C<IR::IROp_UMul>();
|
||||
|
||||
mul(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
mul(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2));
|
||||
}
|
||||
|
||||
DEF_OP(UMull) {
|
||||
auto Op = IROp->C<IR::IROp_UMull>();
|
||||
umull(GetReg(Node).X(), GetReg(Op->Src1.ID()).W(), GetReg(Op->Src2.ID()).W());
|
||||
umull(GetReg(Node).X(), GetReg(Op->Src1).W(), GetReg(Op->Src2).W());
|
||||
}
|
||||
|
||||
DEF_OP(SMull) {
|
||||
auto Op = IROp->C<IR::IROp_SMull>();
|
||||
smull(GetReg(Node).X(), GetReg(Op->Src1.ID()).W(), GetReg(Op->Src2.ID()).W());
|
||||
smull(GetReg(Node).X(), GetReg(Op->Src1).W(), GetReg(Op->Src2).W());
|
||||
}
|
||||
|
||||
DEF_OP(Div) {
|
||||
@@ -412,8 +410,8 @@ DEF_OP(Div) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
@@ -441,8 +439,8 @@ DEF_OP(UDiv) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
@@ -469,8 +467,8 @@ DEF_OP(Rem) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
@@ -498,8 +496,8 @@ DEF_OP(URem) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
@@ -526,8 +524,8 @@ DEF_OP(MulH) {
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
sxtw(TMP1, Src1.W());
|
||||
@@ -546,8 +544,8 @@ DEF_OP(UMulH) {
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
uxtw(ARMEmitter::Size::i64Bit, TMP1, Src1);
|
||||
@@ -562,13 +560,13 @@ DEF_OP(UMulH) {
|
||||
DEF_OP(Orlshl) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshl>();
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(ConvertSize(IROp), Dst, Src1, Const << Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
orr(ConvertSize(IROp), Dst, Src1, Src2, ARMEmitter::ShiftType::LSL, Op->BitShift);
|
||||
}
|
||||
}
|
||||
@@ -577,13 +575,13 @@ DEF_OP(Orlshr) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshr>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(ConvertSize(IROp), Dst, Src1, Const >> Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
orr(ConvertSize(IROp), Dst, Src1, Src2, ARMEmitter::ShiftType::LSR, Op->BitShift);
|
||||
}
|
||||
}
|
||||
@@ -592,9 +590,9 @@ DEF_OP(Ornror) {
|
||||
auto Op = IROp->C<IR::IROp_Ornror>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
orn(ConvertSize(IROp), Dst, Src1, Src2, ARMEmitter::ShiftType::ROR, Op->BitShift);
|
||||
}
|
||||
|
||||
@@ -605,14 +603,14 @@ DEF_OP(AndWithFlags) {
|
||||
|
||||
uint64_t Const;
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
// See TestNZ
|
||||
if (OpSize < IR::OpSize::i32Bit) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
if (Src1 != Src2) {
|
||||
and_(EmitSize, Dst, Src1, Src2);
|
||||
@@ -627,7 +625,7 @@ DEF_OP(AndWithFlags) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ands(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
ands(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
@@ -636,13 +634,13 @@ DEF_OP(AndWithFlags) {
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
|
||||
eor(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
eor(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(XornShift) {
|
||||
auto Op = IROp->C<IR::IROp_XornShift>();
|
||||
|
||||
eon(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
eon(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
@@ -651,7 +649,7 @@ DEF_OP(Ashr) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -663,7 +661,7 @@ DEF_OP(Ashr) {
|
||||
ubfx(EmitSize, Dst, Dst, 0, IR::OpSizeAsBits(OpSize));
|
||||
}
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
asrv(EmitSize, Dst, Src1, Src2);
|
||||
} else {
|
||||
@@ -680,10 +678,10 @@ DEF_OP(ShiftFlags) {
|
||||
const auto EmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto PFOutput = GetReg(Node);
|
||||
const auto PFInput = GetReg(Op->PFInput.ID());
|
||||
const auto Dst = GetReg(Op->Result.ID());
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto PFInput = GetReg(Op->PFInput);
|
||||
const auto Dst = GetReg(Op->Result);
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
bool PFBlocked = (PFOutput == Dst) || (PFOutput == Src1) || (PFOutput == Src2);
|
||||
const auto PFTemp = PFBlocked ? TMP4 : PFOutput;
|
||||
@@ -774,8 +772,8 @@ DEF_OP(ShiftFlags) {
|
||||
|
||||
DEF_OP(RotateFlags) {
|
||||
auto Op = IROp->C<IR::IROp_RotateFlags>();
|
||||
const auto Result = GetReg(Op->Result.ID());
|
||||
const auto Shift = GetReg(Op->Shift.ID());
|
||||
const auto Result = GetReg(Op->Result);
|
||||
const auto Shift = GetReg(Op->Shift);
|
||||
const bool Left = Op->Left;
|
||||
const auto EmitSize = Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
@@ -819,8 +817,8 @@ DEF_OP(RotateFlags) {
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
const auto Lower = GetReg(Op->Lower.ID());
|
||||
const auto Upper = GetReg(Op->Upper);
|
||||
const auto Lower = GetReg(Op->Lower);
|
||||
|
||||
extr(ConvertSize48(IROp), Dst, Upper, Lower, Op->LSB);
|
||||
}
|
||||
@@ -832,8 +830,8 @@ DEF_OP(PDep) {
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
// We can't clobber these
|
||||
const auto OrigInput = GetReg(Op->Input.ID());
|
||||
const auto OrigMask = GetReg(Op->Mask.ID());
|
||||
const auto OrigInput = GetReg(Op->Input);
|
||||
const auto OrigMask = GetReg(Op->Mask);
|
||||
|
||||
if (CTX->HostFeatures.SupportsSVEBitPerm) {
|
||||
// SVE added support for PDEP but it needs to be done in a vector register.
|
||||
@@ -907,8 +905,8 @@ DEF_OP(PExt) {
|
||||
const auto OpSizeBitsM1 = IR::OpSizeAsBits(OpSize) - 1;
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Input = GetReg(Op->Input.ID());
|
||||
const auto Mask = GetReg(Op->Mask.ID());
|
||||
const auto Input = GetReg(Op->Input);
|
||||
const auto Mask = GetReg(Op->Mask);
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
if (CTX->HostFeatures.SupportsSVEBitPerm) {
|
||||
@@ -963,9 +961,9 @@ DEF_OP(LDiv) {
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
const auto Lower = GetReg(Op->Lower.ID());
|
||||
const auto Divisor = GetReg(Op->Divisor.ID());
|
||||
const auto Upper = GetReg(Op->Upper);
|
||||
const auto Lower = GetReg(Op->Lower);
|
||||
const auto Divisor = GetReg(Op->Divisor);
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
@@ -1033,9 +1031,9 @@ DEF_OP(LUDiv) {
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
const auto Lower = GetReg(Op->Lower.ID());
|
||||
const auto Divisor = GetReg(Op->Divisor.ID());
|
||||
const auto Upper = GetReg(Op->Upper);
|
||||
const auto Lower = GetReg(Op->Lower);
|
||||
const auto Divisor = GetReg(Op->Divisor);
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64=
|
||||
@@ -1097,9 +1095,9 @@ DEF_OP(LRem) {
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
const auto Lower = GetReg(Op->Lower.ID());
|
||||
const auto Divisor = GetReg(Op->Divisor.ID());
|
||||
const auto Upper = GetReg(Op->Upper);
|
||||
const auto Lower = GetReg(Op->Lower);
|
||||
const auto Divisor = GetReg(Op->Divisor);
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
@@ -1171,9 +1169,9 @@ DEF_OP(LURem) {
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
const auto Lower = GetReg(Op->Lower.ID());
|
||||
const auto Divisor = GetReg(Op->Divisor.ID());
|
||||
const auto Upper = GetReg(Op->Upper);
|
||||
const auto Lower = GetReg(Op->Lower);
|
||||
const auto Divisor = GetReg(Op->Divisor);
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
@@ -1238,7 +1236,7 @@ DEF_OP(Not) {
|
||||
auto Op = IROp->C<IR::IROp_Not>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
mvn(ConvertSize48(IROp), Dst, Src);
|
||||
}
|
||||
@@ -1248,7 +1246,7 @@ DEF_OP(Popcount) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
@@ -1285,7 +1283,7 @@ DEF_OP(FindLSB) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
// We assume the source is nonzero, so we can just rbit+clz without worrying
|
||||
// about upper garbage for smaller types.
|
||||
@@ -1302,7 +1300,7 @@ DEF_OP(FindMSB) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
movz(ARMEmitter::Size::i64Bit, TMP1, IR::OpSizeAsBits(OpSize) - 1);
|
||||
|
||||
@@ -1325,7 +1323,7 @@ DEF_OP(FindTrailingZeroes) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
|
||||
@@ -1350,7 +1348,7 @@ DEF_OP(CountLeadingZeroes) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
// Expressing as lsl+orr+clz clears away any garbage in the upper bits
|
||||
@@ -1372,7 +1370,7 @@ DEF_OP(Rev) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
rev(EmitSize, Dst, Src);
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
@@ -1385,8 +1383,8 @@ DEF_OP(Bfi) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto SrcDst = GetReg(Op->Dest);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
if (Dst == SrcDst) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
@@ -1417,8 +1415,8 @@ DEF_OP(Bfxil) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto SrcDst = GetReg(Op->Dest);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
if (Dst == SrcDst) {
|
||||
// If Dst and SrcDst match then this turns in to a single instruction.
|
||||
@@ -1443,7 +1441,7 @@ DEF_OP(Bfe) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
if (Op->lsb == 0 && Op->Width == 32) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
@@ -1458,7 +1456,7 @@ DEF_OP(Bfe) {
|
||||
DEF_OP(Sbfe) {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
@@ -1472,18 +1470,18 @@ DEF_OP(Select) {
|
||||
uint64_t Const;
|
||||
auto cc = MapCC(Op->Cond);
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
if (IsGPR(Op->Cmp1)) {
|
||||
const auto Src1 = GetReg(Op->Cmp1);
|
||||
|
||||
if (IsInlineConstant(Op->Cmp2, &Const)) {
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
const auto Src2 = GetReg(Op->Cmp2);
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
} else if (IsFPR(Op->Cmp1)) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1);
|
||||
const auto Src2 = GetVReg(Op->Cmp2);
|
||||
fcmp(Op->CompareSize == IR::OpSize::i64Bit ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
|
||||
@@ -1508,7 +1506,7 @@ DEF_OP(Select) {
|
||||
cset(EmitSize, Dst, cc);
|
||||
}
|
||||
} else {
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetReg(Op->FalseVal.ID()), cc);
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal), GetReg(Op->FalseVal), cc);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1537,7 +1535,7 @@ DEF_OP(NZCVSelect) {
|
||||
cset(EmitSize, Dst, cc);
|
||||
}
|
||||
} else {
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), cc);
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal), GetZeroableReg(Op->FalseVal), cc);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1546,13 +1544,13 @@ DEF_OP(NZCVSelectV) {
|
||||
|
||||
auto cc = MapCC(Op->Cond);
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
fcsel(SubRegSize.Scalar, GetVReg(Node), GetVReg(Op->TrueVal.ID()), GetVReg(Op->FalseVal.ID()), cc);
|
||||
fcsel(SubRegSize.Scalar, GetVReg(Node), GetVReg(Op->TrueVal), GetVReg(Op->FalseVal), cc);
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectIncrement) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectIncrement>();
|
||||
|
||||
csinc(ConvertSize(IROp), GetReg(Node), GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), MapCC(Op->Cond));
|
||||
csinc(ConvertSize(IROp), GetReg(Node), GetReg(Op->TrueVal), GetZeroableReg(Op->FalseVal), MapCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
@@ -1568,7 +1566,7 @@ DEF_OP(VExtractToGPR) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
const auto PerformMove = [&](const ARMEmitter::VRegister reg, int index) {
|
||||
switch (OpSize) {
|
||||
@@ -1617,7 +1615,7 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar);
|
||||
|
||||
if (Op->SrcElementSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.D());
|
||||
@@ -1630,7 +1628,7 @@ DEF_OP(Float_ToGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar);
|
||||
|
||||
if (Op->SrcElementSize == IR::OpSize::i64Bit) {
|
||||
frinti(VTMP1.D(), Src.D());
|
||||
@@ -1645,12 +1643,10 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
const auto EmitSubSize = Op->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1);
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2);
|
||||
|
||||
fcmp(EmitSubSize, Scalar1, Scalar2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -10,18 +10,17 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_A_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
auto Expected0 = GetReg(Op->ExpectedLo.ID());
|
||||
auto Expected1 = GetReg(Op->ExpectedHi.ID());
|
||||
auto Desired0 = GetReg(Op->DesiredLo.ID());
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Dst0 = GetReg(Op->OutLo);
|
||||
auto Dst1 = GetReg(Op->OutHi);
|
||||
auto Expected0 = GetReg(Op->ExpectedLo);
|
||||
auto Expected1 = GetReg(Op->ExpectedHi);
|
||||
auto Desired0 = GetReg(Op->DesiredLo);
|
||||
auto Desired1 = GetReg(Op->DesiredHi);
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
@@ -98,9 +97,9 @@ DEF_OP(CAS) {
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
|
||||
auto Expected = GetReg(Op->Expected.ID());
|
||||
auto Desired = GetReg(Op->Desired.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Expected = GetReg(Op->Expected);
|
||||
auto Desired = GetReg(Op->Desired);
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
@@ -144,8 +143,8 @@ DEF_OP(AtomicAdd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
staddl(SubEmitSize, Src, MemSrc);
|
||||
@@ -164,8 +163,8 @@ DEF_OP(AtomicSub) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
@@ -185,8 +184,8 @@ DEF_OP(AtomicAnd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
@@ -206,8 +205,8 @@ DEF_OP(AtomicCLR) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
@@ -226,8 +225,8 @@ DEF_OP(AtomicOr) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stsetl(SubEmitSize, Src, MemSrc);
|
||||
@@ -246,8 +245,8 @@ DEF_OP(AtomicXor) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
@@ -266,7 +265,7 @@ DEF_OP(AtomicNeg) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
@@ -284,8 +283,8 @@ DEF_OP(AtomicSwap) {
|
||||
"d CAS "
|
||||
"size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
@@ -310,8 +309,8 @@ DEF_OP(AtomicFetchAdd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -331,8 +330,8 @@ DEF_OP(AtomicFetchSub) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
@@ -353,8 +352,8 @@ DEF_OP(AtomicFetchAnd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
@@ -375,8 +374,8 @@ DEF_OP(AtomicFetchCLR) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -396,8 +395,8 @@ DEF_OP(AtomicFetchOr) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -417,8 +416,8 @@ DEF_OP(AtomicFetchXor) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -438,7 +437,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
@@ -452,7 +451,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
DEF_OP(TelemetrySetValue) {
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
auto Op = IROp->C<IR::IROp_TelemetrySetValue>();
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.TelemetryValueAddresses[Op->TelemetryValueIndex]));
|
||||
|
||||
@@ -473,5 +472,4 @@ DEF_OP(TelemetrySetValue) {
|
||||
#endif
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -18,7 +18,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
@@ -82,7 +81,7 @@ DEF_OP(ExitFunction) {
|
||||
} else {
|
||||
|
||||
ARMEmitter::ForwardLabel FullLookup;
|
||||
auto RipReg = GetReg(Op->NewRIP.ID());
|
||||
auto RipReg = GetReg(Op->NewRIP);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
@@ -109,9 +108,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
const auto Target = Op->TargetBlock;
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target.ID()).first->second;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
@@ -125,10 +124,10 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
auto Reg = GetReg(Op->Cmp1);
|
||||
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
@@ -184,7 +183,7 @@ DEF_OP(Syscall) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
str(GetReg(Op->Header.Args[i].ID()).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj));
|
||||
@@ -244,7 +243,7 @@ DEF_OP(InlineSyscall) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
@@ -274,7 +273,7 @@ DEF_OP(InlineSyscall) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
@@ -292,7 +291,7 @@ DEF_OP(InlineSyscall) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i].ID()));
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -325,7 +324,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr));
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
@@ -412,8 +411,8 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf));
|
||||
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
@@ -446,10 +445,10 @@ DEF_OP(CPUID) {
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be 4xi32 scalars
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX.ID()), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX.ID()), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP2, 32, 32);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX), TMP2, 32, 32);
|
||||
}
|
||||
|
||||
DEF_OP(XGetBV) {
|
||||
@@ -458,7 +457,7 @@ DEF_OP(XGetBV) {
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function));
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
@@ -479,9 +478,8 @@ DEF_OP(XGetBV) {
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP1, 32, 32);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX), TMP1, 32, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -23,8 +22,8 @@ DEF_OP(VInsGPR) {
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto DestVector = GetVReg(Op->DestVector);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(ElementSize);
|
||||
@@ -88,7 +87,7 @@ DEF_OP(VInsGPR) {
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
auto Src = GetReg(Op->Src);
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
@@ -109,8 +108,8 @@ DEF_OP(VLoadTwoGPRs) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadTwoGPRs>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcLower = GetReg(Op->Lower.ID());
|
||||
const auto SrcUpper = GetReg(Op->Upper.ID());
|
||||
const auto SrcLower = GetReg(Op->Lower);
|
||||
const auto SrcUpper = GetReg(Op->Upper);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcLower);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcUpper, true);
|
||||
}
|
||||
@@ -120,7 +119,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
@@ -141,7 +140,7 @@ DEF_OP(Float_FromGPR_S) {
|
||||
const uint16_t Conv = (ElementSize << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
auto Src = GetReg(Op->Src);
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- int32_t
|
||||
@@ -179,7 +178,7 @@ DEF_OP(Float_FToF) {
|
||||
const uint16_t Conv = (IR::OpSizeToSize(Op->Header.ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetVReg(Op->Scalar.ID());
|
||||
auto Src = GetVReg(Op->Scalar);
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- Float
|
||||
@@ -220,7 +219,7 @@ DEF_OP(Vector_SToF) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
scvtf(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
@@ -253,7 +252,7 @@ DEF_OP(Vector_FToZS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
@@ -286,7 +285,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
@@ -294,7 +293,7 @@ DEF_OP(Vector_FToS) {
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
@@ -317,7 +316,7 @@ DEF_OP(Vector_FToF) {
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Curiously, FCVTLT and FCVTNT have no bottom variants,
|
||||
@@ -381,7 +380,7 @@ DEF_OP(VFCVTL2) {
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
fcvtl2(SubEmitSize, Dst.D(), Vector.D());
|
||||
}
|
||||
@@ -392,8 +391,8 @@ DEF_OP(VFCVTN2) {
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
auto Lower = VectorLower;
|
||||
if (Dst != VectorLower) {
|
||||
@@ -418,7 +417,7 @@ DEF_OP(Vector_FToI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -479,7 +478,7 @@ DEF_OP(Vector_FToISized) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFRINTTS, "Need FRINTTS for Vector_FToISized");
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (ElementSize == IROp->Size) {
|
||||
// See above
|
||||
@@ -533,7 +532,7 @@ DEF_OP(Vector_F64ToI32) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto Mask = Is256Bit ? PRED_TMP_32B.Merging() : PRED_TMP_16B.Merging();
|
||||
// First step is to round the f64 values to integrals (frint*)
|
||||
@@ -583,5 +582,4 @@ DEF_OP(Vector_F64ToI32) {
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,11 +8,10 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(VAESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector.ID()));
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector));
|
||||
}
|
||||
|
||||
DEF_OP(VAESEnc) {
|
||||
@@ -20,9 +19,9 @@ DEF_OP(VAESEnc) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -45,9 +44,9 @@ DEF_OP(VAESEncLast) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -68,9 +67,9 @@ DEF_OP(VAESDec) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -93,9 +92,9 @@ DEF_OP(VAESDecLast) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -114,9 +113,9 @@ DEF_OP(VAESDecLast) {
|
||||
DEF_OP(VAESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
const auto Swizzle = GetVReg(Op->KeyGenTBLSwizzle.ID());
|
||||
auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Swizzle = GetVReg(Op->KeyGenTBLSwizzle);
|
||||
auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
if (Dst == ZeroReg) {
|
||||
// Seriously? ZeroReg ended up being the destination register?
|
||||
@@ -148,8 +147,8 @@ DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i8Bit: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
@@ -164,7 +163,7 @@ DEF_OP(VSha1H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
@@ -173,9 +172,9 @@ DEF_OP(VSha1C) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1C>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
@@ -193,9 +192,9 @@ DEF_OP(VSha1M) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1M>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
@@ -213,9 +212,9 @@ DEF_OP(VSha1P) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1P>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
@@ -233,8 +232,8 @@ DEF_OP(VSha1SU1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1SU1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1su1(Dst, Src2);
|
||||
@@ -252,9 +251,9 @@ DEF_OP(VSha256H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h(Dst, Src2, Src3);
|
||||
@@ -272,9 +271,9 @@ DEF_OP(VSha256H2) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H2>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
@@ -292,8 +291,8 @@ DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256su0(Dst, Src2);
|
||||
@@ -308,8 +307,8 @@ DEF_OP(VSha256U1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (Dst != Src1 && Dst != Src2) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
@@ -326,8 +325,8 @@ DEF_OP(PCLMUL) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -346,5 +345,4 @@ DEF_OP(PCLMUL) {
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -40,10 +40,6 @@ $end_info$
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace {
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
@@ -80,7 +76,7 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -125,7 +121,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.S(), Src1.S());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -143,7 +139,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -162,7 +158,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// tmp2 (x1/x11): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetReg(IROp->Args[0]);
|
||||
|
||||
// Need to sign or zero extend this for the dispatcher handler.
|
||||
if (Info.ABI == FABI_F80_I16_I16_PTR) {
|
||||
@@ -186,7 +182,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -204,7 +200,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -222,7 +218,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): vector source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -241,8 +237,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp2 (v1/v17): vector source 2
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto Src2 = GetVReg(IROp->Args[1]);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
@@ -262,7 +258,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -280,7 +276,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -298,7 +294,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -317,8 +313,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp2 (v1/v17): vector source 2
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto Src2 = GetVReg(IROp->Args[1]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
|
||||
@@ -337,7 +333,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): vector source 1
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -356,8 +352,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp2 (v1/v17): vector source 2
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto Src2 = GetVReg(IROp->Args[1]);
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
@@ -385,16 +381,16 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
|
||||
stp<ARMEmitter::IndexType::PRE>(TMP1, ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX);
|
||||
const auto SrcRDX = GetReg(Op->RDX);
|
||||
const auto Control = Op->Control;
|
||||
|
||||
mov(TMP1, SrcRAX.X());
|
||||
mov(TMP2, SrcRDX.X());
|
||||
movz(ARMEmitter::Size::i32Bit, TMP3, Control);
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Src1 = GetVReg(Op->LHS);
|
||||
const auto Src2 = GetVReg(Op->RHS);
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
@@ -414,8 +410,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Src1 = GetVReg(Op->LHS);
|
||||
const auto Src2 = GetVReg(Op->RHS);
|
||||
const auto Control = Op->Control;
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
@@ -454,6 +450,8 @@ static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Co
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto Lock = Thread->LookupCache->AcquireLock();
|
||||
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
uintptr_t HostCode {};
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
@@ -493,17 +491,18 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::NodeID Node) {}
|
||||
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
: CPUBackend(*ctx, Thread)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
, CTX {ctx}
|
||||
, TempAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -550,8 +549,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (ParanoidTSO()) {
|
||||
@@ -570,16 +569,24 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
if (WNode.IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
|
||||
@@ -594,6 +601,10 @@ bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
if (WNode.IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
@@ -612,22 +623,6 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsFPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
@@ -689,23 +684,23 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData,
|
||||
bool CheckTF) {
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
uint32_t BufferRange = 0x100 + SSACount * 24;
|
||||
|
||||
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
|
||||
// This minimizes lock contention of CodeBufferWriteMutex.
|
||||
auto TempCodeBuffer = TempAllocator.ReownOrClaimBuffer(BufferRange);
|
||||
SetBuffer(TempCodeBuffer, BufferRange);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
@@ -748,7 +743,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
EmitInterruptChecks(CheckTF);
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
SpillSlots = IR->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
@@ -785,18 +780,17 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
|
||||
|
||||
#define IROP_DISPATCH_DISPATCH
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef REGISTER_OP
|
||||
|
||||
default: Op_Unhandled(IROp, ID); break;
|
||||
default: Op_Unhandled(IROp, CodeNode); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -878,6 +872,49 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
{
|
||||
auto CodeBufferLock = std::unique_lock {CodeBuffers.CodeBufferWriteMutex};
|
||||
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
LOGMAN_THROW_A_FMT(TempSize <= BufferRange, "Exceeded bounds of temporary buffer ({:#x} vs {:#x})", TempSize, BufferRange);
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->Size);
|
||||
SetCursorOffset(AlignUp(CodeBuffers.LatestOffset, 16));
|
||||
if ((GetCursorOffset() + TempSize) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
Align16B();
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
|
||||
// Adjust host addresses
|
||||
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
CodeData.BlockBegin += Delta;
|
||||
CodeData.BlockEntry += Delta;
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
|
||||
TempAllocator.DelayedDisownBuffer();
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeOnlySize);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
|
||||
@@ -38,9 +38,8 @@ public:
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode
|
||||
CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) override;
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
@@ -65,10 +64,10 @@ private:
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
@@ -81,9 +80,17 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
ARMEmitter::Register GetReg(IR::Ref Node) const {
|
||||
return GetReg(IR::PhysicalRegister(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::OrderedNodeWrapper Wrap) const {
|
||||
return GetReg(IR::PhysicalRegister(Wrap));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
@@ -96,15 +103,18 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
ARMEmitter::VRegister GetVReg(IR::Ref Node) const {
|
||||
return GetVReg(IR::PhysicalRegister(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
ARMEmitter::VRegister GetVReg(IR::OrderedNodeWrapper Wrap) const {
|
||||
return GetVReg(IR::PhysicalRegister(Wrap));
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
|
||||
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -114,7 +124,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "Only valid constant");
|
||||
return ARMEmitter::Reg::zr;
|
||||
} else {
|
||||
return GetReg(Src.ID());
|
||||
return GetReg(Src);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,9 +234,34 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
bool IsFPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
bool IsGPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::Ref Node) {
|
||||
return IsGPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::Ref Node) {
|
||||
return IsFPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
@@ -258,7 +293,6 @@ private:
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
const IR::RegisterAllocationData* RAData {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
void ResetStack();
|
||||
@@ -319,7 +353,7 @@ private:
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots {};
|
||||
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node);
|
||||
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::Ref Node);
|
||||
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
@@ -346,7 +380,7 @@ private:
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
@@ -363,6 +397,8 @@ private:
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
|
||||
@@ -15,7 +15,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
@@ -53,8 +52,8 @@ DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), STATE, Op->Offset); break;
|
||||
@@ -62,8 +61,8 @@ DEF_OP(LoadContextPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetVReg(Op->OutValue1);
|
||||
const auto Dst2 = GetVReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), STATE, Op->Offset); break;
|
||||
@@ -89,7 +88,7 @@ DEF_OP(StoreContext) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, STATE, Op->Offset); break;
|
||||
@@ -120,8 +119,8 @@ DEF_OP(StoreContextPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
const auto Src1 = GetVReg(Op->Value1);
|
||||
const auto Src2 = GetVReg(Op->Value2);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), STATE, Op->Offset); break;
|
||||
@@ -137,11 +136,8 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Op->Reg];
|
||||
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
@@ -150,12 +146,10 @@ DEF_OP(LoadRegister) {
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
} else {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
} else {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
@@ -188,26 +182,21 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Reg];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, reg, GetReg(Op->Value));
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "reg out of range");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
const auto host = GetVReg(Op->Value);
|
||||
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
} else {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
} else {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
@@ -217,7 +206,7 @@ DEF_OP(StoreRegister) {
|
||||
DEF_OP(StorePF) {
|
||||
const auto Op = IROp->C<IR::IROp_StorePF>();
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 2];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetReg(Op->Value);
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
@@ -228,7 +217,7 @@ DEF_OP(StorePF) {
|
||||
DEF_OP(StoreAF) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreAF>();
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 1];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetReg(Op->Value);
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
@@ -240,7 +229,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = GetReg(Op->Index.ID());
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -303,10 +292,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = GetReg(Op->Index.ID());
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Value = GetReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -328,7 +317,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride); break;
|
||||
}
|
||||
} else {
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto Value = GetVReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -371,7 +360,7 @@ DEF_OP(SpillRegister) {
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
if (SlotOffset > LSByteMaxUnsignedOffset) {
|
||||
@@ -412,7 +401,7 @@ DEF_OP(SpillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
@@ -552,7 +541,7 @@ DEF_OP(LoadNZCV) {
|
||||
DEF_OP(StoreNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_StoreNZCV>();
|
||||
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value));
|
||||
}
|
||||
|
||||
DEF_OP(LoadDF) {
|
||||
@@ -575,7 +564,7 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
if (IsInlineConstant(Offset, &Const)) {
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
@@ -609,7 +598,7 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const);
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
@@ -676,7 +665,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
// optional extension or shift as part of their behavior.
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset.ID());
|
||||
const auto RegOffset = GetReg(Offset);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
}
|
||||
|
||||
@@ -684,7 +673,7 @@ DEF_OP(LoadMem) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -719,11 +708,11 @@ DEF_OP(LoadMem) {
|
||||
|
||||
DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), Addr, Op->Offset); break;
|
||||
@@ -731,8 +720,8 @@ DEF_OP(LoadMemPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetVReg(Op->OutValue1);
|
||||
const auto Dst2 = GetVReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), Addr, Op->Offset); break;
|
||||
@@ -747,7 +736,7 @@ DEF_OP(LoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
@@ -844,8 +833,8 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -946,9 +935,9 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto RegData = GetVReg(Op->Data.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto RegData = GetVReg(Op->Data);
|
||||
const auto MaskReg = GetVReg(Op->Mask);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto MemDst = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
@@ -1171,13 +1160,13 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
const auto IncomingDst = GetVReg(Op->Incoming);
|
||||
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase.ID())) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask);
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase)) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow);
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh =
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh.ID())) : std::nullopt;
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
|
||||
@@ -1206,7 +1195,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
if (BaseAddr.has_value() || OffsetScale != 1) {
|
||||
ARMEmitter::Register AddrReg = TMP1;
|
||||
if (BaseAddr.has_value()) {
|
||||
AddrReg = GetReg(Op->AddrBase.ID());
|
||||
AddrReg = GetReg(Op->AddrBase);
|
||||
} else {
|
||||
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
|
||||
@@ -1255,13 +1244,13 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
/// - Matches VGATHERQPS/VPGATHERQD behaviour!
|
||||
const auto OffsetScale = Op->OffsetScale;
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
const auto IncomingDst = GetVReg(Op->Incoming);
|
||||
|
||||
const auto MaskReg = GetVReg(Op->MaskReg.ID());
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase.ID())) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow.ID());
|
||||
const auto MaskReg = GetVReg(Op->MaskReg);
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase)) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow);
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh =
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh.ID())) : std::nullopt;
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
|
||||
@@ -1330,8 +1319,8 @@ DEF_OP(VLoadVectorElement) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DstSrc = GetVReg(Op->DstSrc.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto DstSrc = GetVReg(Op->DstSrc);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
@@ -1367,8 +1356,8 @@ DEF_OP(VStoreVectorElement) {
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetVReg(Op->Value);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
@@ -1403,7 +1392,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemReg = GetReg(Op->Address.ID());
|
||||
const auto MemReg = GetReg(Op->Address);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
@@ -1444,8 +1433,8 @@ DEF_OP(VBroadcastFromMem) {
|
||||
DEF_OP(Push) {
|
||||
const auto Op = IROp->C<IR::IROp_Push>();
|
||||
const auto ValueSize = IR::OpSizeToSize(Op->ValueSize);
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
const auto AddrSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value);
|
||||
const auto AddrSrc = GetReg(Op->Addr);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
bool NeedsMoveAfterwards = false;
|
||||
@@ -1526,11 +1515,34 @@ DEF_OP(Push) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PushTwo) {
|
||||
const auto Op = IROp->C<IR::IROp_PushTwo>();
|
||||
const auto ValueSize = IR::OpSizeToSize(Op->ValueSize);
|
||||
auto Src1 = GetReg(Op->Value1);
|
||||
auto Src2 = GetReg(Op->Value2);
|
||||
const auto Dst = GetReg(Op->Addr);
|
||||
|
||||
switch (ValueSize) {
|
||||
case 4: {
|
||||
stp<ARMEmitter::IndexType::PRE>(Src1.W(), Src2.W(), Dst, -2 * ValueSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
stp<ARMEmitter::IndexType::PRE>(Src1.X(), Src2.X(), Dst, -2 * ValueSize);
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ValueSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Pop) {
|
||||
const auto Op = IROp->C<IR::IROp_Pop>();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto Addr = GetReg(Op->InoutAddr.ID());
|
||||
const auto Dst = GetReg(Op->OutValue.ID());
|
||||
const auto Addr = GetReg(Op->InoutAddr);
|
||||
const auto Dst = GetReg(Op->OutValue);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Dst != Addr, "Invalid");
|
||||
|
||||
@@ -1558,11 +1570,42 @@ DEF_OP(Pop) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PopTwo) {
|
||||
const auto Op = IROp->C<IR::IROp_PopTwo>();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto Addr = GetReg(Op->InoutAddr);
|
||||
auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
// ldp x, x is invalid. Explicitly discard the first destination to encode.
|
||||
if (Dst1 == Dst2) {
|
||||
Dst1 = ARMEmitter::Reg::zr;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Dst1 != Addr && Dst2 != Addr, "Invalid");
|
||||
LOGMAN_THROW_A_FMT(Dst1 != Dst2, "Invalid");
|
||||
|
||||
switch (Size) {
|
||||
case 4: {
|
||||
ldp<ARMEmitter::IndexType::POST>(Dst1.W(), Dst2.W(), Addr, 2 * Size);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ldp<ARMEmitter::IndexType::POST>(Dst1.X(), Dst2.X(), Addr, 2 * Size);
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Op->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -1575,7 +1618,7 @@ DEF_OP(StoreMem) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -1615,8 +1658,8 @@ DEF_OP(StoreMemX87SVEOptPredicate) {
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "StoreMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
const auto RegData = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto RegData = GetVReg(Op->Value);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -1644,7 +1687,7 @@ DEF_OP(LoadMemX87SVEOptPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemX87SVEOptPredicate>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Predicate = PRED_X87_SVEOPT;
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "LoadMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
@@ -1674,7 +1717,7 @@ DEF_OP(LoadMemX87SVEOptPredicate) {
|
||||
DEF_OP(StoreMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
@@ -1685,8 +1728,8 @@ DEF_OP(StoreMemPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
const auto Src1 = GetVReg(Op->Value1);
|
||||
const auto Src2 = GetVReg(Op->Value2);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), Addr, Op->Offset); break;
|
||||
@@ -1701,7 +1744,7 @@ DEF_OP(StoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
@@ -1751,7 +1794,7 @@ DEF_OP(StoreMemTSO) {
|
||||
// Half-Barrier.
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
@@ -1782,16 +1825,16 @@ DEF_OP(MemSet) {
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Value = GetZeroableReg(Op->Value);
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Length = GetReg(Op->Length);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
DirectionReg = GetReg(Op->Direction);
|
||||
}
|
||||
|
||||
// If Direction > 0 then:
|
||||
@@ -1808,7 +1851,7 @@ DEF_OP(MemSet) {
|
||||
if (Op->Prefix.IsInvalid()) {
|
||||
mov(TMP2, MemReg.X());
|
||||
} else {
|
||||
const auto Prefix = GetReg(Op->Prefix.ID());
|
||||
const auto Prefix = GetReg(Op->Prefix);
|
||||
add(TMP2, Prefix.X(), MemReg.X());
|
||||
}
|
||||
|
||||
@@ -1971,19 +2014,19 @@ DEF_OP(MemCpy) {
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemRegDest = GetReg(Op->Dest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->Src.ID());
|
||||
const auto MemRegDest = GetReg(Op->Dest);
|
||||
const auto MemRegSrc = GetReg(Op->Src);
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Length = GetReg(Op->Length);
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
DirectionReg = GetReg(Op->Direction);
|
||||
}
|
||||
|
||||
auto Dst0 = GetReg(Op->OutDstAddress.ID());
|
||||
auto Dst1 = GetReg(Op->OutSrcAddress.ID());
|
||||
auto Dst0 = GetReg(Op->OutDstAddress);
|
||||
auto Dst1 = GetReg(Op->OutSrcAddress);
|
||||
// If Direction > 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
@@ -2241,7 +2284,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -2329,7 +2372,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
@@ -2362,7 +2405,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
@@ -2416,7 +2459,7 @@ DEF_OP(CacheLineClear) {
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
@@ -2445,7 +2488,7 @@ DEF_OP(CacheLineClean) {
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClean>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Clean dcache only
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
@@ -2463,7 +2506,7 @@ DEF_OP(CacheLineClean) {
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
@@ -2483,7 +2526,7 @@ DEF_OP(CacheLineZero) {
|
||||
|
||||
DEF_OP(Prefetch) {
|
||||
auto Op = IROp->C<IR::IROp_Prefetch>();
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Access size is only ever handled as 8-byte. Even though it is accesssed as a cacheline.
|
||||
const auto MemSrc = GenerateMemOperand(IR::OpSize::i64Bit, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -2527,8 +2570,8 @@ DEF_OP(VStoreNonTemporal) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetVReg(Op->Value);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
@@ -2552,10 +2595,10 @@ DEF_OP(VStoreNonTemporalPair) {
|
||||
[[maybe_unused]] const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "This IR operation only operates at 128-bit wide");
|
||||
|
||||
const auto ValueLow = GetVReg(Op->ValueLow.ID());
|
||||
const auto ValueHigh = GetVReg(Op->ValueHigh.ID());
|
||||
const auto ValueLow = GetVReg(Op->ValueLow);
|
||||
const auto ValueHigh = GetVReg(Op->ValueHigh);
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
stnp(ValueLow.Q(), ValueHigh.Q(), MemReg, Offset);
|
||||
@@ -2570,7 +2613,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
@@ -2587,5 +2630,4 @@ DEF_OP(VLoadNonTemporal) {
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -16,11 +16,6 @@ $end_info$
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AllocateGPR) {}
|
||||
DEF_OP(AllocateGPRAfter) {}
|
||||
DEF_OP(AllocateFPR) {}
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
@@ -102,8 +97,8 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg(Op->RoundMode.ID());
|
||||
auto MXCSR = GetReg(Op->MXCSR.ID());
|
||||
auto Src = GetReg(Op->RoundMode);
|
||||
auto MXCSR = GetReg(Op->MXCSR);
|
||||
|
||||
// As above, setup the rounding flags in [31:30]
|
||||
rbit(ARMEmitter::Size::i32Bit, TMP2, Src);
|
||||
@@ -161,7 +156,7 @@ DEF_OP(PushRoundingMode) {
|
||||
|
||||
DEF_OP(PopRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_PopRoundingMode>();
|
||||
msr(ARMEmitter::SystemRegister::FPCR, GetReg(Op->FPCR.ID()));
|
||||
msr(ARMEmitter::SystemRegister::FPCR, GetReg(Op->FPCR));
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
@@ -170,17 +165,17 @@ DEF_OP(Print) {
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
if (IsGPR(Op->Value)) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
|
||||
} else {
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value.ID()), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value.ID()), true);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value), true);
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
if (IsGPR(Op->Value)) {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
@@ -271,5 +266,4 @@ DEF_OP(Yield) {
|
||||
yield();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,26 +8,19 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(Copy) {
|
||||
auto Op = IROp->C<IR::IROp_Copy>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source));
|
||||
}
|
||||
|
||||
DEF_OP(RMWHandle) {
|
||||
auto Op = IROp->C<IR::IROp_RMWHandle>();
|
||||
auto Dest = GetReg(Node);
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Dest != Src) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dest, Src);
|
||||
}
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A.ID()), B = GetReg(Op->B.ID());
|
||||
auto A = GetReg(Op->A), B = GetReg(Op->B);
|
||||
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
@@ -39,5 +32,4 @@ DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -11,7 +11,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
#define DEF_UNOP(FEXOp, ARMOp, ScalarCase) \
|
||||
DEF_OP(FEXOp) { \
|
||||
@@ -24,7 +23,7 @@ namespace FEXCore::CPU {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
const auto Src = GetVReg(Op->Vector); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
@@ -45,8 +44,8 @@ namespace FEXCore::CPU {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
@@ -65,8 +64,8 @@ namespace FEXCore::CPU {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
@@ -85,8 +84,8 @@ namespace FEXCore::CPU {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID()); \
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID()); \
|
||||
const auto VectorLower = GetVReg(Op->VectorLower); \
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), VectorLower.Z(), VectorUpper.Z()); \
|
||||
@@ -110,7 +109,7 @@ namespace FEXCore::CPU {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
const auto Src = GetVReg(Op->Vector); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
@@ -149,8 +148,8 @@ namespace FEXCore::CPU {
|
||||
const auto IsScalar = ElementSize == OpSize; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
@@ -188,8 +187,8 @@ namespace FEXCore::CPU {
|
||||
}; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2); \
|
||||
}
|
||||
@@ -211,10 +210,10 @@ namespace FEXCore::CPU {
|
||||
}; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Upper = GetVReg(Op->Upper.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Addend = GetVReg(Op->Addend.ID()); \
|
||||
const auto Upper = GetVReg(Op->Upper); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
const auto Addend = GetVReg(Op->Addend); \
|
||||
\
|
||||
VFScalarFMAOperation(IROp->Size, ElementSize, ScalarEmit, Dst, Upper, Vector1, Vector2, Addend); \
|
||||
}
|
||||
@@ -458,8 +457,8 @@ DEF_OP(VFMinScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -484,8 +483,8 @@ DEF_OP(VFMaxScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -503,8 +502,8 @@ DEF_OP(VFSqrtScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -541,8 +540,8 @@ DEF_OP(VFRSqrtScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -578,8 +577,8 @@ DEF_OP(VFRecpScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -624,8 +623,8 @@ DEF_OP(VFToFScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -656,8 +655,8 @@ DEF_OP(VSToFVectorInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// Claim the element size is 8-bytes.
|
||||
// Might be scalar 8-byte (cvtsi2ss xmm0, rax)
|
||||
@@ -729,8 +728,8 @@ DEF_OP(VSToFGPRInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto GPR = GetReg(Op->Src.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto GPR = GetReg(Op->Src);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector, GPR);
|
||||
}
|
||||
@@ -756,8 +755,8 @@ DEF_OP(VFToIScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -916,8 +915,8 @@ DEF_OP(VFCMPScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Funcs[FEXCore::ToUnderlying(Op->Op)], Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -1033,7 +1032,7 @@ DEF_OP(VMov) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Source = GetVReg(Op->Source.ID());
|
||||
const auto Source = GetVReg(Op->Source);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -1087,8 +1086,8 @@ DEF_OP(VAddP) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1125,7 +1124,7 @@ DEF_OP(VFAddV) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit || OpSize == IR::OpSize::i256Bit, "Only AVX and SSE size supported");
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -1156,7 +1155,7 @@ DEF_OP(VAddV) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// SVE doesn't have an equivalent ADDV instruction, so we make do
|
||||
@@ -1192,7 +1191,7 @@ DEF_OP(VUMinV) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B;
|
||||
@@ -1213,7 +1212,7 @@ DEF_OP(VUMaxV) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B;
|
||||
@@ -1233,8 +1232,8 @@ DEF_OP(VURAvg) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -1267,8 +1266,8 @@ DEF_OP(VFAddP) {
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1302,8 +1301,8 @@ DEF_OP(VFDiv) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -1358,8 +1357,8 @@ DEF_OP(VFMin) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// NOTE: We don't directly use FMIN** here for any of the implementations,
|
||||
// because it has undesirable NaN handling behavior (it sets
|
||||
@@ -1431,8 +1430,8 @@ DEF_OP(VFMax) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// NOTE: See VFMin implementation for reasons why we
|
||||
// don't just use FMAX/FMIN for these implementations.
|
||||
@@ -1489,7 +1488,7 @@ DEF_OP(VFRecp) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1559,7 +1558,7 @@ DEF_OP(VFRecpPrecision) {
|
||||
const auto IsScalar = OpSize == ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (IsScalar) {
|
||||
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
|
||||
@@ -1598,7 +1597,7 @@ DEF_OP(VFRSqrt) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1669,7 +1668,7 @@ DEF_OP(VFRSqrtPrecision) {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (IsScalar) {
|
||||
if (HostSupportsRPRES) {
|
||||
@@ -1707,7 +1706,7 @@ DEF_OP(VNot) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
not_(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
@@ -1726,8 +1725,8 @@ DEF_OP(VUMin) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1775,8 +1774,8 @@ DEF_OP(VSMin) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1824,8 +1823,8 @@ DEF_OP(VUMax) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1873,8 +1872,8 @@ DEF_OP(VSMax) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1921,9 +1920,9 @@ DEF_OP(VBSL) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorFalse = GetVReg(Op->VectorFalse.ID());
|
||||
const auto VectorTrue = GetVReg(Op->VectorTrue.ID());
|
||||
const auto VectorMask = GetVReg(Op->VectorMask.ID());
|
||||
const auto VectorFalse = GetVReg(Op->VectorFalse);
|
||||
const auto VectorTrue = GetVReg(Op->VectorTrue);
|
||||
const auto VectorMask = GetVReg(Op->VectorMask);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// NOTE: Slight parameter difference from ASIMD
|
||||
@@ -1990,8 +1989,8 @@ DEF_OP(VCMPEQ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
@@ -2030,7 +2029,7 @@ DEF_OP(VCMPEQZ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2072,8 +2071,8 @@ DEF_OP(VCMPGT) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2112,7 +2111,7 @@ DEF_OP(VCMPGTZ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2151,7 +2150,7 @@ DEF_OP(VCMPLTZ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2190,8 +2189,8 @@ DEF_OP(VFCMPEQ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2229,8 +2228,8 @@ DEF_OP(VFCMPNEQ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2270,8 +2269,8 @@ DEF_OP(VFCMPLT) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2309,8 +2308,8 @@ DEF_OP(VFCMPGT) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2348,8 +2347,8 @@ DEF_OP(VFCMPLE) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2387,8 +2386,8 @@ DEF_OP(VFCMPORD) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2438,8 +2437,8 @@ DEF_OP(VFCMPUNO) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2489,8 +2488,8 @@ DEF_OP(VUShl) {
|
||||
const auto MaxShift = IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -2545,8 +2544,8 @@ DEF_OP(VUShr) {
|
||||
const auto MaxShift = IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -2605,8 +2604,8 @@ DEF_OP(VSShr) {
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2660,8 +2659,8 @@ DEF_OP(VUShlS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2689,8 +2688,8 @@ DEF_OP(VUShrS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2720,8 +2719,8 @@ DEF_OP(VUShrSWide) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2786,8 +2785,8 @@ DEF_OP(VSShrSWide) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2852,8 +2851,8 @@ DEF_OP(VUShlSWide) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2916,8 +2915,8 @@ DEF_OP(VSShrS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2950,8 +2949,8 @@ DEF_OP(VInsElement) {
|
||||
const uint32_t SrcIdx = Op->SrcIdx;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcVector = GetVReg(Op->SrcVector.ID());
|
||||
auto Reg = GetVReg(Op->DestVector.ID());
|
||||
const auto SrcVector = GetVReg(Op->SrcVector);
|
||||
auto Reg = GetVReg(Op->DestVector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Broadcast our source value across a temporary,
|
||||
@@ -3029,7 +3028,7 @@ DEF_OP(VDupElement) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(SubRegSize, Dst.Z(), Vector.Z(), Index);
|
||||
@@ -3050,8 +3049,8 @@ DEF_OP(VExtr) {
|
||||
|
||||
// AArch64 ext op has bit arrangement as [Vm:Vn] so arguments need to be swapped
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto UpperBits = GetVReg(Op->VectorLower.ID());
|
||||
auto LowerBits = GetVReg(Op->VectorUpper.ID());
|
||||
auto UpperBits = GetVReg(Op->VectorLower);
|
||||
auto LowerBits = GetVReg(Op->VectorUpper);
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
auto Index = Op->Index;
|
||||
@@ -3101,7 +3100,7 @@ DEF_OP(VUShrI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (BitShift >= IR::OpSizeAsBits(ElementSize)) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
@@ -3143,8 +3142,8 @@ DEF_OP(VUShraI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto DestVector = GetVReg(Op->DestVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst == DestVector) {
|
||||
@@ -3187,7 +3186,7 @@ DEF_OP(VSShrI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -3226,7 +3225,7 @@ DEF_OP(VShlI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (BitShift >= IR::OpSizeAsBits(ElementSize)) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
@@ -3268,7 +3267,7 @@ DEF_OP(VUShrNI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
@@ -3292,8 +3291,8 @@ DEF_OP(VUShrNI2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_16B;
|
||||
@@ -3329,7 +3328,7 @@ DEF_OP(VSXTL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
sunpklo(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3347,7 +3346,7 @@ DEF_OP(VSXTL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
sunpkhi(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3365,7 +3364,7 @@ DEF_OP(VSSHLL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
LOGMAN_THROW_A_FMT(BitShift < IR::OpSizeAsBits(IROp->ElementSize / 2), "Bitshift size too large for source element size: {} < {}",
|
||||
BitShift, IR::OpSizeAsBits(IROp->ElementSize / 2));
|
||||
@@ -3387,7 +3386,7 @@ DEF_OP(VSSHLL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
LOGMAN_THROW_A_FMT(BitShift < IR::OpSizeAsBits(IROp->ElementSize / 2), "Bitshift size too large for source element size: {} < {}",
|
||||
BitShift, IR::OpSizeAsBits(IROp->ElementSize / 2));
|
||||
@@ -3409,7 +3408,7 @@ DEF_OP(VUXTL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
uunpklo(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3427,7 +3426,7 @@ DEF_OP(VUXTL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
uunpkhi(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3445,7 +3444,7 @@ DEF_OP(VSQXTN) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Note that SVE SQXTNB and SQXTNT are a tad different
|
||||
@@ -3497,8 +3496,8 @@ DEF_OP(VSQXTN2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// We use the 16 byte mask due to how SPLICE works. We only
|
||||
@@ -3541,8 +3540,8 @@ DEF_OP(VSQXTNPair) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// This combines the SVE versions of VSQXTN/VSQXTN2.
|
||||
@@ -3585,7 +3584,7 @@ DEF_OP(VSQXTUN) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
sqxtunb(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3604,8 +3603,8 @@ DEF_OP(VSQXTUN2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// NOTE: See VSQXTN2 implementation for an in-depth explanation
|
||||
@@ -3650,8 +3649,8 @@ DEF_OP(VSQXTUNPair) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// This combines the SVE versions of VSQXTUN/VSQXTUN2.
|
||||
@@ -3695,7 +3694,7 @@ DEF_OP(VSRSHR) {
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -3725,7 +3724,7 @@ DEF_OP(VSQSHL) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -3755,8 +3754,8 @@ DEF_OP(VMul) {
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
mul(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3774,8 +3773,8 @@ DEF_OP(VUMull) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
umullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3795,8 +3794,8 @@ DEF_OP(VSMull) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
smullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3816,8 +3815,8 @@ DEF_OP(VUMull2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
umullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3837,8 +3836,8 @@ DEF_OP(VSMull2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
smullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3861,8 +3860,8 @@ DEF_OP(VUMulH) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
const auto SubRegSizeLarger = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -3912,8 +3911,8 @@ DEF_OP(VSMulH) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
const auto SubRegSizeLarger = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -3960,8 +3959,8 @@ DEF_OP(VUABDL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
@@ -3986,8 +3985,8 @@ DEF_OP(VUABDL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
@@ -4008,8 +4007,8 @@ DEF_OP(VTBL1) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
const auto VectorTable = GetVReg(Op->VectorTable.ID());
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices);
|
||||
const auto VectorTable = GetVReg(Op->VectorTable);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i64Bit: {
|
||||
@@ -4035,9 +4034,9 @@ DEF_OP(VTBL2) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
auto VectorTable1 = GetVReg(Op->VectorTable1.ID());
|
||||
auto VectorTable2 = GetVReg(Op->VectorTable2.ID());
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices);
|
||||
auto VectorTable1 = GetVReg(Op->VectorTable1);
|
||||
auto VectorTable2 = GetVReg(Op->VectorTable2);
|
||||
|
||||
if (!ARMEmitter::AreVectorsSequential(VectorTable1, VectorTable2)) {
|
||||
// Vector registers aren't sequential, need to move to temporaries.
|
||||
@@ -4079,9 +4078,9 @@ DEF_OP(VTBX1) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorSrcDst = GetVReg(Op->VectorSrcDst.ID());
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
const auto VectorTable = GetVReg(Op->VectorTable.ID());
|
||||
const auto VectorSrcDst = GetVReg(Op->VectorSrcDst);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices);
|
||||
const auto VectorTable = GetVReg(Op->VectorTable);
|
||||
|
||||
if (Dst != VectorSrcDst) {
|
||||
switch (OpSize) {
|
||||
@@ -4136,7 +4135,7 @@ DEF_OP(VRev32) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit, "Invalid size");
|
||||
const auto SubRegSize = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
@@ -4175,7 +4174,7 @@ DEF_OP(VRev64) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4213,8 +4212,8 @@ DEF_OP(VFCADD) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Rotate == 90 || Op->Rotate == 270, "Invalidate Rotate");
|
||||
const auto Rotate = Op->Rotate == 90 ? ARMEmitter::Rotation::ROTATE_90 : ARMEmitter::Rotation::ROTATE_270;
|
||||
@@ -4260,9 +4259,9 @@ DEF_OP(VFMLA) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4327,9 +4326,9 @@ DEF_OP(VFMLS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4417,9 +4416,9 @@ DEF_OP(VFNMLA) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4487,9 +4486,9 @@ DEF_OP(VFNMLS) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4568,8 +4567,8 @@ DEF_OP(VFCopySign) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
ARMEmitter::VRegister Magnitude = GetVReg(Op->Vector1.ID());
|
||||
ARMEmitter::VRegister Sign = GetVReg(Op->Vector2.ID());
|
||||
ARMEmitter::VRegister Magnitude = GetVReg(Op->Vector1);
|
||||
ARMEmitter::VRegister Sign = GetVReg(Op->Vector2);
|
||||
|
||||
// We don't assign explicity to Dst but Dst and Magniture are tied to the same register.
|
||||
// Similar in semantics to C's copysignf.
|
||||
@@ -4586,6 +4585,4 @@ DEF_OP(VFCopySign) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -14,14 +14,17 @@ $end_info$
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
namespace FEXCore {
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()}
|
||||
, ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
GuestToHostMap::GuestToHostMap()
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
}
|
||||
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -62,18 +65,30 @@ LookupCache::~LookupCache() {
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -16,6 +16,86 @@
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
struct GuestToHostMap {
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
struct LockToken {
|
||||
std::lock_guard<std::recursive_mutex> Lock;
|
||||
};
|
||||
|
||||
[[nodiscard]]
|
||||
LockToken AcquireLock() {
|
||||
return LockToken {std::lock_guard {WriteLock}};
|
||||
}
|
||||
|
||||
struct BlockLinkTag {
|
||||
uint64_t GuestDestination;
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink;
|
||||
|
||||
bool operator<(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination) {
|
||||
return true;
|
||||
} else if (GuestDestination == other.GuestDestination) {
|
||||
return HostLink < other.HostLink;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Use a monotonic buffer resource to allocate both the std::pmr::map and its members.
|
||||
// This allows us to quickly clear the block link map by clearing the monotonic allocator.
|
||||
// If we had allocated the block link map without the MBR, then clearing the map would require slowly
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
if (HostCode == BlockList.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
BlockList.erase(Address);
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache(const LockToken&);
|
||||
};
|
||||
|
||||
class LookupCache {
|
||||
public:
|
||||
struct LookupCacheEntry {
|
||||
@@ -26,6 +106,13 @@ public:
|
||||
LookupCache(FEXCore::Context::ContextImpl* CTX);
|
||||
~LookupCache();
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -34,7 +121,7 @@ public:
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
@@ -56,29 +143,30 @@ public:
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto& CodePage = CodePages[CurrentPage];
|
||||
rv |= CodePage.size() == 0;
|
||||
rv |= CodePage.empty();
|
||||
CodePage.push_back(Address);
|
||||
}
|
||||
|
||||
@@ -87,10 +175,9 @@ public:
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_A_FMT(Inserted, "Duplicate block mapping added");
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
@@ -99,19 +186,13 @@ public:
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
BlockList.erase(Address);
|
||||
Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -141,13 +222,13 @@ public:
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
auto lk = Shared->AcquireLock();
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
@@ -169,7 +250,9 @@ public:
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
std::recursive_mutex WriteLock;
|
||||
auto AcquireLock() {
|
||||
return Shared->AcquireLock();
|
||||
}
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
@@ -226,34 +309,6 @@ private:
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
|
||||
struct BlockLinkTag {
|
||||
uint64_t GuestDestination;
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink;
|
||||
|
||||
bool operator<(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination) {
|
||||
return true;
|
||||
} else if (GuestDestination == other.GuestDestination) {
|
||||
return HostLink < other.HostLink;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Use a monotonic buffer resource to allocate both the std::pmr::map and its members.
|
||||
// This allows us to quickly clear the block link map by clearing the monotonic allocator.
|
||||
// If we had allocated the block link map without the MBR, then clearing the map would require slowly
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
|
||||
@@ -355,68 +355,29 @@ void OpDispatchBuilder::SALCOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Push(Size, Src);
|
||||
FlushRegisterCache();
|
||||
Push(Size, LoadSource(GPRClass, Op, Op->Src[0], Op->Flags));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Push(Size, Src);
|
||||
FlushRegisterCache();
|
||||
Push(Size, LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHAOp(OpcodeArgs) {
|
||||
// 32bit only
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
Ref OldSP = _Copy(LoadGPRRegister(X86State::REG_RSP));
|
||||
|
||||
// PUSHA order:
|
||||
// Tmp = SP
|
||||
// push EAX
|
||||
// push ECX
|
||||
// push EDX
|
||||
// push EBX
|
||||
// push Tmp
|
||||
// push EBP
|
||||
// push ESI
|
||||
// push EDI
|
||||
|
||||
Ref Src {};
|
||||
Ref NewSP = OldSP;
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RAX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RCX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RDX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RBX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
// Push old-sp
|
||||
NewSP = _Push(GPRSize, Size, OldSP, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RBP);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RSI);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RDI);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP, OpSize::i32Bit);
|
||||
FlushRegisterCache();
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RAX));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RCX));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RDX));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RBX));
|
||||
Push(Size, OldSP);
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RBP));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RSI));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RDI));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
@@ -1083,9 +1044,9 @@ void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
auto Size = GetSrcSize(Op);
|
||||
Ref Upper = _Sbfe(OpSize::i64Bit, 1, Size * 8 - 1, Src);
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Size = OpSizeFromSrc(Op);
|
||||
Ref Upper = _Sbfe(std::max(OpSize::i32Bit, Size), 1, GetSrcBitSize(Op) - 1, Src);
|
||||
|
||||
StoreResult(GPRClass, Op, Upper, OpSize::iInvalid);
|
||||
}
|
||||
@@ -1185,10 +1146,11 @@ void OpDispatchBuilder::FLAGControlOp(OpcodeArgs) {
|
||||
SetCFInverted(_Constant(0));
|
||||
break;
|
||||
case 0xFC: // CLD
|
||||
SetRFLAG(_Constant(0), FEXCore::X86State::RFLAG_DF_RAW_LOC);
|
||||
// Transformed
|
||||
StoreDF(_Constant(1));
|
||||
break;
|
||||
case 0xFD: // STD
|
||||
SetRFLAG(_Constant(1), FEXCore::X86State::RFLAG_DF_RAW_LOC);
|
||||
StoreDF(_Constant(-1));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1481,10 +1443,10 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
Ref Res {};
|
||||
if (Size < 32) {
|
||||
Ref ShiftLeft = _Constant(Shift);
|
||||
auto ShiftRight = _Constant(Size - Shift);
|
||||
auto ShiftRight = Size - Shift;
|
||||
|
||||
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, ShiftLeft);
|
||||
auto Tmp2 = _Lshr(OpSize::i32Bit, Src, ShiftRight);
|
||||
Ref Tmp2 = ShiftRight ? _Lshr(OpSize::i32Bit, Src, _Constant(ShiftRight)) : Src;
|
||||
|
||||
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
|
||||
} else {
|
||||
@@ -1606,15 +1568,14 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
// things tighter for 8-bit later in the function.
|
||||
uint64_t Mask = Size == 8 ? 7 : (Size == 64 ? 0x3F : 0x1F);
|
||||
|
||||
Ref Src, UnmaskedSrc;
|
||||
ArithRef UnmaskedSrc;
|
||||
if (Is1Bit || IsImmediate) {
|
||||
UnmaskedConst = LoadConstantShift(Op, Is1Bit);
|
||||
UnmaskedSrc = _Constant(UnmaskedConst);
|
||||
Src = _Constant(UnmaskedConst & Mask);
|
||||
UnmaskedSrc = ARef(UnmaskedConst);
|
||||
} else {
|
||||
UnmaskedSrc = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = _And(OpSize::i64Bit, UnmaskedSrc, _InlineConstant(Mask));
|
||||
UnmaskedSrc = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
}
|
||||
auto Src = UnmaskedSrc.And(Mask);
|
||||
|
||||
// We fill the upper bits so we allow garbage on load.
|
||||
auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
@@ -1626,18 +1587,18 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
}
|
||||
|
||||
// To rotate 64-bits left, right-rotate by (64 - Shift) = -Shift mod 64.
|
||||
auto Res = _Ror(OpSize, Dest, Left ? _Neg(OpSize, Src) : Src);
|
||||
auto Res = _Ror(OpSize, Dest, (Left ? Src.Neg() : Src).Ref());
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
if (Is1Bit || IsImmediate) {
|
||||
if (UnmaskedConst) {
|
||||
if (UnmaskedSrc.C) {
|
||||
// Extract the last bit shifted in to CF
|
||||
SetCFDirect(Res, Left ? 0 : Size - 1, true);
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
// For ROL, OF is the LSB and MSB XOR'd together.
|
||||
// OF is architecturally only defined for 1-bit rotate.
|
||||
if (UnmaskedConst == 1) {
|
||||
if (UnmaskedSrc.C == 1) {
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, Left ? Size - 1 : 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Left ? 0 : Size - 2, true);
|
||||
}
|
||||
@@ -1648,10 +1609,10 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
|
||||
// We deferred the masking for 8-bit to the flag section, do it here.
|
||||
if (Size == 8) {
|
||||
Src = _And(OpSize::i64Bit, UnmaskedSrc, _InlineConstant(0x1F));
|
||||
Src = UnmaskedSrc.And(0x1F);
|
||||
}
|
||||
|
||||
_RotateFlags(OpSizeFromSrc(Op), Res, Src, Left);
|
||||
_RotateFlags(OpSizeFromSrc(Op), Res, Src.Ref(), Left);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1662,7 +1623,7 @@ void OpDispatchBuilder::ANDNBMIOp(OpcodeArgs) {
|
||||
auto Dest = _Andn(OpSizeFromSrc(Op), Src2, Src1);
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
CalculateFlags_Logical(OpSizeFromSrc(Op), Dest, Src1, Src2);
|
||||
CalculateFlags_Logical(OpSizeFromSrc(Op), Dest);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) {
|
||||
@@ -2109,15 +2070,15 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src, [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -2182,27 +2143,23 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
// Entire bitfield has been setup. Just extract the 8 or 16bits we need.
|
||||
// 64-bit shift used because we want to rotate in our cascaded upper bits
|
||||
// rather than zeroes.
|
||||
Ref Res = _Lshr(OpSize::i64Bit, Tmp, Src);
|
||||
Ref Res = _Lshr(OpSize::i64Bit, Tmp, Src.Ref());
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
uint64_t SrcConst = 0;
|
||||
bool IsSrcConst = IsValueConstant(WrapNode(Src), &SrcConst);
|
||||
SrcConst &= 0x1f;
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source. 32-bit Lshr masks the
|
||||
// same as x86, but if we constant fold we must mask ourselves.
|
||||
if (IsSrcConst) {
|
||||
SetCFDirect(Tmp, SrcConst - 1, true);
|
||||
if (Src.IsConstant) {
|
||||
SetCFDirect(Tmp, (Src.C & 0x1f) - 1, true);
|
||||
} else {
|
||||
auto One = _Constant(OpSizeFromSrc(Op), 1);
|
||||
auto NewCF = _Lshr(OpSize::i32Bit, Tmp, _Sub(OpSize::i32Bit, Src, One));
|
||||
auto NewCF = _Lshr(OpSize::i32Bit, Tmp, _Sub(OpSize::i32Bit, Src.Ref(), One));
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
}
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (!IsSrcConst || SrcConst == 1) {
|
||||
if (!Src.IsConstant || Src.C == 1) {
|
||||
auto NewOF = _XorShift(OpSize::i32Bit, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Size - 2, true);
|
||||
}
|
||||
@@ -2329,15 +2286,15 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src, [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
@@ -2360,20 +2317,19 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
// Shift 1 more bit that expected to get our result
|
||||
// Shifting to the right will now behave like a rotate to the left
|
||||
// Which we emulate with a _Ror
|
||||
Ref Res = _Ror(OpSize::i64Bit, Tmp, _Neg(OpSize::i32Bit, Src));
|
||||
Ref Res = _Ror(OpSize::i64Bit, Tmp, Src.Neg().Ref());
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
// Our new CF is now at the bit position that we are shifting
|
||||
// Either 0 if CF hasn't changed (CF is living in bit 0)
|
||||
// or higher
|
||||
auto NewCF = _Ror(OpSize::i64Bit, Tmp, _Sub(OpSize::i64Bit, _Constant(63), Src));
|
||||
auto NewCF = _Ror(OpSize::i64Bit, Tmp, Src.Presub(63).Ref());
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
|
||||
// OF is the XOR of the NewCF and the MSB of the result
|
||||
// Only defined for 1-bit rotates.
|
||||
uint64_t SrcConst;
|
||||
if (!IsValueConstant(WrapNode(Src), &SrcConst) || SrcConst == 1) {
|
||||
if (!Src.IsConstant || Src.C == 1) {
|
||||
auto NewOF = _XorShift(OpSize::i64Bit, NewCF, Res, ShiftType::LSR, Size - 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
|
||||
}
|
||||
@@ -2382,21 +2338,19 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Ref Value;
|
||||
Ref Src {};
|
||||
ArithRef Src;
|
||||
bool IsNonconstant = Op->Src[SrcIndex].IsGPR();
|
||||
uint8_t ConstantShift = 0;
|
||||
|
||||
const uint32_t Size = GetDstBitSize(Op);
|
||||
const uint32_t Mask = Size - 1;
|
||||
|
||||
if (IsNonconstant) {
|
||||
// Because we mask explicitly with And/Bfe/Sbfe after, we can allow garbage here.
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = ARef(LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
} else {
|
||||
// Can only be an immediate
|
||||
// Masked by operand size
|
||||
ConstantShift = Op->Src[SrcIndex].Data.Literal.Value & Mask;
|
||||
Src = _Constant(ConstantShift);
|
||||
Src = ARef(Op->Src[SrcIndex].Data.Literal.Value & Mask);
|
||||
}
|
||||
|
||||
if (Op->Dest.IsGPR()) {
|
||||
@@ -2408,7 +2362,8 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
// Get the bit selection from the src. We need to mask for 8/16-bit, but
|
||||
// rely on the implicit masking of Lshr for native sizes.
|
||||
unsigned LshrSize = std::max<uint8_t>(IR::OpSizeToSize(OpSize::i32Bit), Size / 8);
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : _And(OpSize::i64Bit, Src, _Constant(Mask));
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : Src.And(Mask);
|
||||
auto LshrOpSize = IR::SizeToOpSize(LshrSize);
|
||||
|
||||
// OF/SF/AF/PF undefined. ZF must be preserved. We choose to preserve OF/SF
|
||||
// too since we just use an rmif to insert into CF directly. We could
|
||||
@@ -2418,10 +2373,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
// can reuse the invert.
|
||||
if (Action != BTAction::BTComplement) {
|
||||
if (IsNonconstant) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect);
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect.Ref());
|
||||
}
|
||||
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, Src.IsConstant ? Src.C : 0, true);
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
@@ -2432,30 +2387,27 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTClear: {
|
||||
Ref BitMask = _Lshl(IR::SizeToOpSize(LshrSize), _Constant(1), BitSelect);
|
||||
Dest = _Andn(IR::SizeToOpSize(LshrSize), Dest, BitMask);
|
||||
Dest = _Andn(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref());
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
|
||||
case BTAction::BTSet: {
|
||||
Ref BitMask = _Lshl(IR::SizeToOpSize(LshrSize), _Constant(1), BitSelect);
|
||||
Dest = _Or(IR::SizeToOpSize(LshrSize), Dest, BitMask);
|
||||
Dest = _Or(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref());
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
|
||||
case BTAction::BTComplement: {
|
||||
Ref BitMask = _Lshl(IR::SizeToOpSize(LshrSize), _Constant(1), BitSelect);
|
||||
Dest = _Xor(IR::SizeToOpSize(LshrSize), Dest, BitMask);
|
||||
Dest = _Xor(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref());
|
||||
|
||||
if (IsNonconstant) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Dest, BitSelect);
|
||||
Value = _Lshr(LshrOpSize, Dest, BitSelect.Ref());
|
||||
} else {
|
||||
Value = Dest;
|
||||
}
|
||||
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, Src.IsConstant ? Src.C : 0, true);
|
||||
CFInverted = true;
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
@@ -2466,17 +2418,15 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
// Load the address to the memory location
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
// Get the bit selection from the src
|
||||
Ref BitSelect = _Bfe(std::max(OpSize::i32Bit, GetOpSize(Src)), 3, 0, Src);
|
||||
auto BitSelect = Src.Bfe(0, 3);
|
||||
|
||||
// Address is provided as bits we want BYTE offsets
|
||||
// Extract Signed offset
|
||||
Src = _Sbfe(OpSize::i64Bit, Size - 3, 3, Src);
|
||||
Src = Src.Sbfe(3, Size - 3);
|
||||
|
||||
// Get the address offset by shifting out the size of the op (To shift out the bit selection)
|
||||
// Then use that to index in to the memory location by size of op
|
||||
AddressMode Address = {.Base = Dest, .Index = Src, .AddrSize = OpSize::i64Bit};
|
||||
|
||||
ConstantShift = 0;
|
||||
AddressMode Address = {.Base = Dest, .Index = Src.Ref(), .AddrSize = OpSize::i64Bit};
|
||||
|
||||
switch (Action) {
|
||||
case BTAction::BTNone: {
|
||||
@@ -2485,7 +2435,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTClear: {
|
||||
Ref BitMask = _Lshl(OpSize::i64Bit, _Constant(1), BitSelect);
|
||||
Ref BitMask = BitSelect.MaskBit(OpSize::i64Bit).Ref();
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
@@ -2500,7 +2450,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTSet: {
|
||||
Ref BitMask = _Lshl(OpSize::i64Bit, _Constant(1), BitSelect);
|
||||
Ref BitMask = BitSelect.MaskBit(OpSize::i64Bit).Ref();
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
@@ -2515,7 +2465,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTComplement: {
|
||||
Ref BitMask = _Lshl(OpSize::i64Bit, _Constant(1), BitSelect);
|
||||
Ref BitMask = BitSelect.MaskBit(OpSize::i64Bit).Ref();
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
@@ -2531,10 +2481,12 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Value = _Lshr(std::max(OpSize::i32Bit, GetOpSize(Value)), Value, BitSelect);
|
||||
if (!BitSelect.IsDefinitelyZero()) {
|
||||
Value = _Lshr(std::max(OpSize::i32Bit, GetOpSize(Value)), Value, BitSelect.Ref());
|
||||
}
|
||||
|
||||
// OF/SF/ZF/AF/PF undefined.
|
||||
SetCFDirect(Value, ConstantShift, true);
|
||||
SetCFDirect(Value, 0, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2591,7 +2543,7 @@ void OpDispatchBuilder::IMUL2SrcOp(OpcodeArgs) {
|
||||
case OpSize::i8Bit:
|
||||
case OpSize::i16Bit: {
|
||||
Src1 = _Sbfe(OpSize::i64Bit, SizeBits, 0, Src1);
|
||||
Src2 = _Sbfe(OpSize::i64Bit, SizeBits, 0, Src2);
|
||||
Src2 = ARef(Src2).Sbfe(0, SizeBits).Ref();
|
||||
Dest = _Mul(OpSize::i64Bit, Src1, Src2);
|
||||
ResultHigh = _Sbfe(OpSize::i64Bit, SizeBits, SizeBits, Dest);
|
||||
break;
|
||||
@@ -3594,8 +3546,7 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUSHFOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = GetPackedRFLAG();
|
||||
Push(Size, Src);
|
||||
Push(Size, GetPackedRFLAG());
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::POPFOp(OpcodeArgs) {
|
||||
@@ -3933,7 +3884,7 @@ void OpDispatchBuilder::CreateJumpBlocks(const fextl::vector<FEXCore::Frontend::
|
||||
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions) {
|
||||
Entry = RIP;
|
||||
auto IRHeader = _IRHeader(InvalidNode, RIP, 0, NumInstructions);
|
||||
auto IRHeader = _IRHeader(InvalidNode, RIP, 0, NumInstructions, 0, 0);
|
||||
CreateJumpBlocks(Blocks);
|
||||
|
||||
auto Block = GetNewJumpBlock(RIP);
|
||||
@@ -4299,7 +4250,7 @@ void OpDispatchBuilder::StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize
|
||||
Ref Reg = Src;
|
||||
if (Size != GPRSize || Offset != 0) {
|
||||
// Need to do an insert if not automatic size or zero offset.
|
||||
Reg = _Bfi(GPRSize, IR::OpSizeAsBits(Size), Offset, LoadGPRRegister(GPR), Src);
|
||||
Reg = ARef(Reg).BfiInto(LoadGPRRegister(GPR), Offset, IR::OpSizeAsBits(Size));
|
||||
}
|
||||
|
||||
StoreRegister(GPR, false, Reg);
|
||||
@@ -4358,7 +4309,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (GPRSize == OpSize::i64Bit && OpSize == OpSize::i32Bit) {
|
||||
// If the Source IR op is 64 bits, we need to zext the upper bits
|
||||
// For all other sizes, the upper bits are guaranteed to already be zero
|
||||
Ref Value = GetOpSize(Src) == OpSize::i64Bit ? _Bfe(OpSize::i32Bit, 32, 0, Src) : Src;
|
||||
Ref Value = GetOpSize(Src) == OpSize::i64Bit ? ARef(Src).Bfe(0, 32).Ref() : Src;
|
||||
StoreGPRRegister(gpr, Value, GPRSize);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
@@ -4456,18 +4407,22 @@ void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx) {
|
||||
/* On x86, the canonical way to zero a register is XOR with itself... because
|
||||
* modern x86 detects this pattern in hardware. arm64 does not detect this
|
||||
* pattern, we should do it like the x86 hardware would. On arm64, "mov x0,
|
||||
* #0" is faster than "eor x0, x0, x0". Additionally this lets more constant
|
||||
* folding kick in for flags.
|
||||
*/
|
||||
// On x86, the canonical way to zero a register is XOR with itself. Detect and
|
||||
// emit optimal arm64 assembly.
|
||||
if (!DestIsLockedMem(Op) && ALUIROp == FEXCore::IR::IROps::OP_XOR && Op->Dest.IsGPR() && Op->Src[SrcIdx].IsGPR() &&
|
||||
Op->Dest.Data.GPR == Op->Src[SrcIdx].Data.GPR) {
|
||||
|
||||
auto Result = _Constant(0);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
CalculateFlags_Logical(OpSizeFromSrc(Op), Result, Result, Result);
|
||||
// Move 0 into the register
|
||||
StoreResult(GPRClass, Op, _Constant(0), OpSize::iInvalid);
|
||||
|
||||
// Set flags for zero result with inverted carry. We subtract an arbitrary
|
||||
// register from itself to get the zero, since `subs wzr, #0` is not
|
||||
// encodable. This is optimal and works regardless of the opsize.
|
||||
auto Zero = LoadGPR(Op->Dest.Data.GPR.GPR);
|
||||
HandleNZ00Write();
|
||||
InvalidateAF();
|
||||
CalculatePF(_SubWithFlags(OpSize::i32Bit, Zero, Zero));
|
||||
CFInverted = true;
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -4485,7 +4440,8 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
// Try to eliminate the masking after 8/16-bit operations with constants, by
|
||||
// promoting to a full size operation that preserves the upper bits.
|
||||
uint64_t Const;
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits && IsValueConstant(WrapNode(Src), &Const) &&
|
||||
bool IsConst = IsValueConstant(WrapNode(Src), &Const);
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits && IsConst &&
|
||||
(ALUIROp == IR::IROps::OP_XOR || ALUIROp == IR::IROps::OP_OR || ALUIROp == IR::IROps::OP_ANDWITHFLAGS)) {
|
||||
|
||||
RoundedSize = ResultSize = CTX->GetGPROpSize();
|
||||
@@ -4519,8 +4475,15 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
}
|
||||
|
||||
const auto OpSize = RoundedSize;
|
||||
DeriveOp(ALUOp, ALUIROp, _AndWithFlags(OpSize, Dest, Src));
|
||||
Result = ALUOp;
|
||||
uint64_t Mask = Size == OpSize::i64Bit ? ~0ull : ((1ull << IR::OpSizeAsBits(Size)) - 1);
|
||||
if (IsConst && Const == Mask && !DestIsLockedMem(Op) && ALUIROp == IR::IROps::OP_XOR && Size >= OpSize::i32Bit) {
|
||||
Result = _Not(OpSize, Dest);
|
||||
} else if (IsConst && Const == Mask && !DestIsLockedMem(Op) && ALUIROp == IR::IROps::OP_AND) {
|
||||
Result = Dest;
|
||||
} else {
|
||||
DeriveOp(ALUOp, ALUIROp, _AndWithFlags(OpSize, Dest, Src));
|
||||
Result = ALUOp;
|
||||
}
|
||||
|
||||
// Flags set
|
||||
switch (ALUIROp) {
|
||||
@@ -4529,7 +4492,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
case FEXCore::IR::IROps::OP_XOR:
|
||||
case FEXCore::IR::IROps::OP_AND:
|
||||
case FEXCore::IR::IROps::OP_OR: {
|
||||
CalculateFlags_Logical(Size, Result, Dest, Src);
|
||||
CalculateFlags_Logical(Size, Result);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::IROps::OP_ANDWITHFLAGS: {
|
||||
|
||||
@@ -161,7 +161,7 @@ public:
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(NewRIP);
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP);
|
||||
}
|
||||
IRPair<IROp_Break> Break(BreakDefinition Reason) {
|
||||
FlushRegisterCache();
|
||||
@@ -2025,14 +2025,7 @@ private:
|
||||
|
||||
// Returns (DF ? -Size : Size)
|
||||
Ref LoadDir(const unsigned Size) {
|
||||
auto Dir = LoadDF();
|
||||
auto Shift = FEXCore::ilog2(Size);
|
||||
|
||||
if (Shift) {
|
||||
return _Lshl(CTX->GetGPROpSize(), Dir, _Constant(Shift));
|
||||
} else {
|
||||
return Dir;
|
||||
}
|
||||
return ARef(LoadDF()).Lshl(FEXCore::ilog2(Size)).Ref();
|
||||
}
|
||||
|
||||
// Returns DF ? (X - Size) : (X + Size)
|
||||
@@ -2310,7 +2303,7 @@ private:
|
||||
Ref CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
|
||||
void CalculateFlags_ShiftLeft(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
@@ -2321,20 +2314,7 @@ private:
|
||||
void CalculateFlags_ZCNT(IR::OpSize SrcSize, Ref Result);
|
||||
/** @} */
|
||||
|
||||
Ref AndConst(FEXCore::IR::OpSize Size, Ref Node, uint64_t Const) {
|
||||
uint64_t NodeConst;
|
||||
|
||||
if (IsValueConstant(WrapNode(Node), &NodeConst)) {
|
||||
return _Constant(NodeConst & Const);
|
||||
} else {
|
||||
return _And(Size, Node, _Constant(Const));
|
||||
}
|
||||
}
|
||||
|
||||
/** @} */
|
||||
|
||||
Ref GetX87Top();
|
||||
Ref GetX87Tag(Ref Value, Ref AbridgedFTW);
|
||||
void SetX87FTW(Ref FTW);
|
||||
Ref GetX87FTW_Helper();
|
||||
void SetX87Top(Ref Value);
|
||||
@@ -2494,10 +2474,139 @@ private:
|
||||
return Value;
|
||||
}
|
||||
|
||||
Ref VZeroExtendOperand(OpSize Size, X86Tables::DecodedOperand Op, Ref Value) {
|
||||
bool IsMMX = Op.IsGPR() && Op.Data.GPR.GPR >= X86State::REG_MM_0;
|
||||
bool AlreadyExtended = Op.IsGPRDirect() || Op.IsGPRIndirect() || IsMMX;
|
||||
|
||||
return AlreadyExtended ? Value : _VMov(Size, Value);
|
||||
}
|
||||
|
||||
void Push(IR::OpSize Size, Ref Value) {
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
auto NewSP = _Push(CTX->GetGPROpSize(), Size, Value, OldSP);
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
FlushRegisterCache();
|
||||
}
|
||||
|
||||
struct ArithRef {
|
||||
IREmitter* E;
|
||||
bool IsConstant;
|
||||
union {
|
||||
Ref R;
|
||||
uint64_t C;
|
||||
};
|
||||
|
||||
ArithRef() {}
|
||||
|
||||
ArithRef(IREmitter* IREmit, Ref Reference)
|
||||
: E(IREmit)
|
||||
, IsConstant(false)
|
||||
, R(Reference) {}
|
||||
|
||||
ArithRef(IREmitter* IREmit, uint64_t K)
|
||||
: E(IREmit)
|
||||
, IsConstant(true)
|
||||
, C(K) {}
|
||||
|
||||
ArithRef Neg() {
|
||||
return IsConstant ? ArithRef(E, -C) : ArithRef(E, E->_Neg(OpSize::i64Bit, R));
|
||||
}
|
||||
|
||||
ArithRef And(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->_Constant(K)));
|
||||
}
|
||||
|
||||
ArithRef Presub(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->_Sub(OpSize::i64Bit, E->_Constant(K), R));
|
||||
}
|
||||
|
||||
ArithRef Lshl(uint64_t Shift) {
|
||||
if (Shift == 0) {
|
||||
return *this;
|
||||
} else if (IsConstant) {
|
||||
return ArithRef(E, C << Shift);
|
||||
} else {
|
||||
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->_Constant(Shift)));
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef Bfe(unsigned Start, unsigned Size) {
|
||||
if (IsConstant) {
|
||||
return ArithRef(E, (C >> Start) & ((1ull << Size) - 1));
|
||||
} else {
|
||||
return ArithRef(E, E->_Bfe(OpSize::i64Bit, Size, Start, R));
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef Sbfe(unsigned Start, unsigned Size) {
|
||||
if (IsConstant) {
|
||||
uint64_t SourceMask = Size == 64 ? ~0ULL : ((1ULL << Size) - 1);
|
||||
SourceMask <<= Start;
|
||||
|
||||
int64_t NewConstant = (C & SourceMask) >> Start;
|
||||
NewConstant <<= 64 - Size;
|
||||
NewConstant >>= 64 - Size;
|
||||
|
||||
return ArithRef(E, NewConstant);
|
||||
} else {
|
||||
return ArithRef(E, E->_Sbfe(OpSize::i64Bit, Size, Start, R));
|
||||
}
|
||||
}
|
||||
|
||||
Ref BfiInto(Ref Bitfield, unsigned Start, unsigned Size) {
|
||||
if (IsConstant && (Size > 0 && Size < 64)) {
|
||||
uint64_t SourceMask = (1ULL << Size) - 1;
|
||||
uint64_t SourceMaskShifted = SourceMask << Start;
|
||||
|
||||
if (C == 0) {
|
||||
return E->_And(OpSize::i64Bit, Bitfield, E->_InlineConstant(~SourceMaskShifted));
|
||||
} else if (C == SourceMask) {
|
||||
return E->_Or(OpSize::i64Bit, Bitfield, E->_InlineConstant(SourceMaskShifted));
|
||||
}
|
||||
}
|
||||
|
||||
if (IsConstant) {
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->_Constant(C));
|
||||
} else {
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, R);
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef MaskBit(OpSize Size) {
|
||||
if (IsConstant) {
|
||||
uint64_t ShiftMask = Size == OpSize::i64Bit ? 63 : 31;
|
||||
uint64_t Result = 1ull << (C & ShiftMask);
|
||||
if (ShiftMask == 31) {
|
||||
Result &= ((1ull << 32) - 1);
|
||||
}
|
||||
|
||||
return ArithRef(E, Result);
|
||||
} else {
|
||||
return ArithRef(E, E->_Lshl(Size, E->_Constant(1), R));
|
||||
}
|
||||
}
|
||||
|
||||
Ref Ref() {
|
||||
return IsConstant ? E->_Constant(C) : R;
|
||||
}
|
||||
|
||||
bool IsDefinitelyZero() {
|
||||
return IsConstant && C == 0;
|
||||
}
|
||||
};
|
||||
|
||||
ArithRef ARef(Ref R) {
|
||||
uint64_t C;
|
||||
|
||||
if (IsValueConstant(WrapNode(R), &C)) {
|
||||
return ARef(C);
|
||||
} else {
|
||||
return ArithRef(this, R);
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef ARef(uint64_t K) {
|
||||
return ArithRef(this, K);
|
||||
}
|
||||
|
||||
void InstallHostSpecificOpcodeHandlers();
|
||||
|
||||
@@ -845,7 +845,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
|
||||
if (Op->Dest.IsGPR()) {
|
||||
// Zero bits [127:64] as well.
|
||||
Src.Low = _VMov(OpSize::i64Bit, Src.Low);
|
||||
Src.Low = VZeroExtendOperand(OpSize::i64Bit, Op->Src[0], Src.Low);
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i128Bit);
|
||||
Src.High = ZeroVector;
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
|
||||
@@ -228,21 +228,26 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
// We only care about bit 4 in the subsequent XOR. If we'll XOR with 0,
|
||||
// there's no sense XOR'ing at all. If we'll XOR with 1, that's just
|
||||
// inverting.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src2), &Const)) {
|
||||
if (Const & (1u << 4)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Not(OpSize::i32Bit, Src1));
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Src1);
|
||||
}
|
||||
for (unsigned i = 0; i < 2; ++i) {
|
||||
Ref SrcA = i ? Src1 : Src2;
|
||||
Ref SrcB = i ? Src2 : Src1;
|
||||
|
||||
return;
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(SrcA), &Const)) {
|
||||
if (Const & (1u << 4)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Not(OpSize::i32Bit, SrcB));
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(SrcB);
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit. Again 64-bit to avoid masking.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
Ref XorRes = Src1 == Src2 ? _Constant(0) : _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -276,7 +281,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
CFInverted = false;
|
||||
} else {
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
Src2 = ARef(Src2).Bfe(0, IR::OpSizeAsBits(SrcSize)).Ref();
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
@@ -316,7 +321,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src1);
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
Src2 = ARef(Src2).Bfe(0, IR::OpSizeAsBits(SrcSize)).Ref();
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
@@ -426,13 +431,9 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res) {
|
||||
InvalidateAF();
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// SF/ZF/CF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetNZP_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
|
||||
@@ -78,7 +78,7 @@ void OpDispatchBuilder::VMOVAPS_VMOVAPDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
if (Is128Bit && Op->Dest.IsGPR()) {
|
||||
Src = _VMov(OpSize::i128Bit, Src);
|
||||
Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Src, OpSize::iInvalid);
|
||||
}
|
||||
@@ -90,7 +90,7 @@ void OpDispatchBuilder::VMOVUPS_VMOVUPDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
|
||||
|
||||
if (Is128Bit && Op->Dest.IsGPR()) {
|
||||
Src = _VMov(OpSize::i128Bit, Src);
|
||||
Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Src, OpSize::i8Bit);
|
||||
}
|
||||
@@ -706,7 +706,7 @@ void OpDispatchBuilder::MOVQOp(OpcodeArgs, VectorOpType VectorType) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
auto Reg = _VMov(OpSize::i64Bit, Src);
|
||||
auto Reg = VZeroExtendOperand(OpSize::i64Bit, Op->Src[0], Src);
|
||||
StoreXMMRegister_WithAVXInsert(VectorType, gprIndex, Reg);
|
||||
} else {
|
||||
// This is simple, just store the result
|
||||
@@ -762,11 +762,12 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
Ref Tmp = _VExtractToGPR(Size, ElementSize, Src, i);
|
||||
Tmp = _Bfe(ElementSize, 1, IR::OpSizeAsBits(ElementSize) - 1, Tmp);
|
||||
|
||||
// Shift it to the correct location
|
||||
Tmp = _Lshl(ElementSize, Tmp, _Constant(i));
|
||||
|
||||
// Or it with the current value
|
||||
CurrentVal = _Or(OpSize::i64Bit, CurrentVal, Tmp);
|
||||
// Shift it to the correct location and or it with the current value
|
||||
if (i != 0) {
|
||||
CurrentVal = _Orlshl(OpSize::i64Bit, CurrentVal, Tmp, i);
|
||||
} else {
|
||||
CurrentVal = Tmp;
|
||||
}
|
||||
}
|
||||
StoreResult(GPRClass, Op, CurrentVal, OpSize::iInvalid);
|
||||
}
|
||||
@@ -3006,7 +3007,7 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) {
|
||||
if constexpr (ToXMM) {
|
||||
const auto Index = Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0;
|
||||
|
||||
Src = _VMov(OpSize::i128Bit, Src);
|
||||
Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src);
|
||||
StoreXMMRegister(Index, Src);
|
||||
} else {
|
||||
// This is simple, just store the result
|
||||
|
||||
@@ -31,14 +31,6 @@ Ref OpDispatchBuilder::GetX87Top() {
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
Ref RegValid = _And(OpSize::i32Bit, _Lshr(OpSize::i32Bit, AbridgedFTW, Value), _Constant(1));
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref X87Valid = _Constant(static_cast<uint8_t>(FPState::X87Tag::Valid));
|
||||
|
||||
return _Select(FEXCore::IR::COND_EQ, RegValid, _Constant(0), X87Empty, X87Valid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW {};
|
||||
@@ -312,14 +304,25 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
Ref FTW = _Constant(0);
|
||||
// AbridgedFTWIndex has 1-bit per slot (8 slots). Duplicate each bit to get
|
||||
// 2-bits per slot (16-bit result). Duplicating bits is equivalent to
|
||||
// Morton interleaving a number with itself. To interleave efficiently two
|
||||
// bytes, we use the well-known bit twiddling algorithm:
|
||||
//
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = LoadContext(AbridgedFTWIndex);
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, _Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
X = _And(OpSize::i32Bit, X, _Constant(0x33333333));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 1);
|
||||
X = _And(OpSize::i32Bit, X, _Constant(0x55555555));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 1);
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
Ref RegTag = GetX87Tag(_Constant(i), LoadContext(AbridgedFTWIndex));
|
||||
FTW = _Orlshl(OpSize::i32Bit, FTW, RegTag, i * 2);
|
||||
}
|
||||
|
||||
return FTW;
|
||||
// The above sequence sets valid to 11 and empty to 00, so invert to finalize.
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Valid) == 0b00);
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
|
||||
return _Xor(OpSize::i32Bit, X, _Constant(0xffff));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
|
||||
@@ -47,19 +47,12 @@ AOTIRInlineEntry* AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
IR::RegisterAllocationData* AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData*)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView* AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView*)&InlineData[Offset];
|
||||
return (IR::IRListView*)InlineData;
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash,
|
||||
const FEXCore::IR::IRListView& IRList, const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
const FEXCore::IR::IRListView& IRList) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->Offset());
|
||||
|
||||
if (Inserted.second) {
|
||||
@@ -69,8 +62,6 @@ void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t
|
||||
};
|
||||
Stream->Write((const char*)&entry, sizeof(entry));
|
||||
|
||||
RAData->Serialize(*Stream);
|
||||
|
||||
// IRData (inline)
|
||||
IRList.Serialize(*Stream);
|
||||
}
|
||||
@@ -265,9 +256,6 @@ class IRInlineStorage : public IRStorageBase {
|
||||
public:
|
||||
IRInlineStorage(AOTIRInlineEntry& entry)
|
||||
: entry(entry) {}
|
||||
const RegisterAllocationData* RAData() override {
|
||||
return entry.GetRAData();
|
||||
}
|
||||
IRListView GetIRView() override {
|
||||
return entry.GetIRData();
|
||||
}
|
||||
@@ -332,7 +320,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && IR->RAData() && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
if (GeneratedIR && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
@@ -356,7 +344,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->Write(&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView(), IR->RAData());
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView());
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
|
||||
@@ -50,7 +50,6 @@ class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
|
||||
constexpr auto COOKIE_VERSION = [](const char CookieText[4], uint32_t Version) {
|
||||
@@ -75,10 +74,9 @@ struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
|
||||
/* RAData followed by IRData */
|
||||
/* IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData* GetRAData();
|
||||
IR::IRListView* GetIRData();
|
||||
};
|
||||
|
||||
@@ -100,8 +98,7 @@ struct AOTIRCaptureCacheEntry {
|
||||
fextl::unique_ptr<FEXCore::Context::AOTIRWriter> Stream;
|
||||
fextl::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData);
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList);
|
||||
};
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
|
||||
@@ -11,7 +11,6 @@ namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
/**
|
||||
* @brief The IROp_Header is an dynamically sized array
|
||||
@@ -106,8 +105,7 @@ struct NodeID final {
|
||||
*/
|
||||
template<typename Type>
|
||||
struct FEX_PACKED NodeWrapperBase final {
|
||||
// On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [<Base> + <Index> + <imm offset>]
|
||||
// On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter
|
||||
// 32bit or 64bit offset doesn't matter for addressing.
|
||||
// We use uint32_t to be more memory efficient (Cuts our node list size in half)
|
||||
using NodeOffsetType = uint32_t;
|
||||
NodeOffsetType NodeOffset;
|
||||
@@ -141,22 +139,54 @@ struct FEX_PACKED NodeWrapperBase final {
|
||||
return NodeOffset == 0;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsImmediate() const {
|
||||
return NodeOffset & (1u << 31);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsPointer() const {
|
||||
return !IsImmediate();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
Type* GetNode(uintptr_t Base) {
|
||||
LOGMAN_THROW_A_FMT(IsPointer(), "Precondition");
|
||||
return reinterpret_cast<Type*>(Base + NodeOffset);
|
||||
}
|
||||
[[nodiscard]]
|
||||
const Type* GetNode(uintptr_t Base) const {
|
||||
LOGMAN_THROW_A_FMT(IsPointer(), "Precondition");
|
||||
return reinterpret_cast<const Type*>(Base + NodeOffset);
|
||||
}
|
||||
|
||||
void SetOffset(uintptr_t Base, uintptr_t Value) {
|
||||
NodeOffset = Value - Base;
|
||||
LOGMAN_THROW_A_FMT(IsPointer(), "Offsets are within 2GiB range");
|
||||
}
|
||||
|
||||
void SetImmediate(uint32_t Immediate) {
|
||||
LOGMAN_THROW_A_FMT(Immediate < (1u << 31), "Bounded");
|
||||
NodeOffset = Immediate | (1u << 31);
|
||||
LOGMAN_THROW_A_FMT(IsImmediate(), "Encoded above");
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
uint32_t GetImmediate() const {
|
||||
LOGMAN_THROW_A_FMT(IsImmediate(), "Precondition: must be an immediate");
|
||||
return NodeOffset & ~(1u << 31);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
|
||||
[[nodiscard]]
|
||||
static NodeWrapperBase<Type> FromImmediate(uint32_t Immediate) {
|
||||
NodeWrapperBase<Type> A;
|
||||
A.SetImmediate(Immediate);
|
||||
return A;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivially_copyable_v<NodeWrapperBase<OrderedNode>>);
|
||||
@@ -196,6 +226,15 @@ public:
|
||||
OrderedNodeHeader Header;
|
||||
uint32_t NumUses;
|
||||
|
||||
// After RA, the register allocated for the node. This is the register for the
|
||||
// node at the time it is written, even if it is shuffled into other registers
|
||||
// later. In other words, it is the register destination of the instruction
|
||||
// represented by this OrderedNode.
|
||||
//
|
||||
// This is the raw value of a PhysicalRegister data structure.
|
||||
uint8_t Reg;
|
||||
uint8_t Pad[3];
|
||||
|
||||
using value_type = OrderedNodeWrapper;
|
||||
|
||||
OrderedNode() = default;
|
||||
@@ -358,7 +397,7 @@ private:
|
||||
static_assert(std::is_trivially_constructible_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + 2 * sizeof(uint32_t)));
|
||||
|
||||
// This is temporary. We are transitioning away from OrderedNode's in favour of
|
||||
// flat Ref words. To ease porting, we have this typedef. Eventually OrderedNode
|
||||
@@ -726,7 +765,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op);
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData);
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR);
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
|
||||
@@ -169,11 +169,11 @@
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}, i1:$ReadsParity{false}": {
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, u32:$SpillSlots, i1:$PostRA{false}, i1:$HasX87{false}, i1:$ReadsParity{false}": {
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"CodeBlock SSA:$Begin, SSA:$Last": {
|
||||
"CodeBlock SSA:$Begin, SSA:$Last, u32:$ID": {
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
@@ -254,19 +254,22 @@
|
||||
"it cannot use a regular destination too. This ensures RA correctness.",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"If ForPair is set, RA will try to allocate the base of a register pair"],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = AllocateFPR OpSize:#RegisterSize, OpSize:#ElementSize": {
|
||||
"Desc": ["Like AllocateGPR, but for FPR"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = AllocateGPRAfter GPR:$After": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"RA will attempt to allocate to the register after $After.",
|
||||
"It may not succeed."],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = RDRAND i1:$GetReseeded": {
|
||||
"Desc": ["Uses the hardware random number generator to generate a 64bit number",
|
||||
@@ -294,11 +297,11 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction GPR:$NewRIP": {
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "GetOpSize(NewRIP)"
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"Break BreakDefinition:$Reason": {
|
||||
"HasSideEffects": true
|
||||
@@ -500,17 +503,13 @@
|
||||
]
|
||||
},
|
||||
|
||||
"SSA = FillRegister SSA:$OriginalValue, u32:$Slot, RegisterClass:$Class": {
|
||||
"SSA = FillRegister OpSize:#Size, OpSize:#ElementSize, u32:$Slot, RegisterClass:$Class": {
|
||||
"Desc": ["Fills a register from a spill slot",
|
||||
"Spill slots are register allocated and has live ranges calculated to handle slot calculation",
|
||||
"```diff\n- !Don't use this op. It is for RA to handle spilling and filling!\n```",
|
||||
"",
|
||||
"The OriginalValue SSA arg points at the original SSA value spilled, and only exists for",
|
||||
"RA validation purposes"
|
||||
"```diff\n- !Don't use this op. It is for RA to handle spilling and filling!\n```"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($OriginalValue) == $Class"
|
||||
]
|
||||
"DestSize": "Size",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"GPR = LoadNZCV": {
|
||||
@@ -672,11 +671,19 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"PushTwo OpSize:#Size, OpSize:$ValueSize, GPR:$Value1, GPR:$Value2, GPR:$Addr": {
|
||||
"Desc": [
|
||||
"Push two values to the address, incrementing the pointer in the place.",
|
||||
"Fused post-RA so doesn't have a destination."
|
||||
],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = RMWHandle GPR:$Value": {
|
||||
"Desc": [
|
||||
"This is a special move that indicates the result will be poisoned by a non-SSA instruction writing to its result.",
|
||||
"In effect, it serves to prevent invalid optimizations with non-SSA instructions."
|
||||
],
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"HasSideEffects": true,
|
||||
"TiedSource": 0
|
||||
},
|
||||
@@ -688,6 +695,11 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR:$Addr, GPR:$Value1, GPR:$Value2 = PopTwo OpSize:$Size, GPR:$Addr": {
|
||||
"Desc": ["Pop two values from the address. Fused post-RA."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = MemSet i1:$IsAtomic, OpSize:$Size, GPR:$Prefix, GPR:$Addr, GPR:$Value, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 STOS repeat",
|
||||
"Returns the final address that gets generated without the prefix appended."
|
||||
|
||||
@@ -82,7 +82,27 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, const IR::RegisterAllocationData* RAData) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsImmediate()) {
|
||||
auto PhyReg = PhysicalRegister(Arg);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "c"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "invalid"; break;
|
||||
default: *out << "unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg;
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
const auto ArgID = Arg.ID();
|
||||
|
||||
@@ -90,25 +110,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%" << std::dec << ArgID;
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
} else {
|
||||
*out << ")";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
@@ -271,7 +272,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData) {
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
@@ -323,16 +324,16 @@ void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllo
|
||||
|
||||
*out << "%" << std::dec << ID;
|
||||
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
if (!PhyReg.IsInvalid()) {
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(invalid"; break;
|
||||
default: *out << "(unknown"; break;
|
||||
}
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
|
||||
@@ -146,6 +146,10 @@ void IREmitter::RemoveArgUses(Ref Node) {
|
||||
}
|
||||
}
|
||||
|
||||
void IREmitter::RemovePostRA(Ref Node) {
|
||||
Node->Unlink(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
void IREmitter::Remove(Ref Node) {
|
||||
RemoveArgUses(Node);
|
||||
|
||||
@@ -185,27 +189,4 @@ void IREmitter::SetCurrentCodeBlock(Ref Node) {
|
||||
SetWriteCursor(Node->Op(DualListData.DataBegin())->CW<IROp_CodeBlock>()->Begin.GetNode(DualListData.ListBegin()));
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceWithConstant(Ref Node, uint64_t Value) {
|
||||
auto Header = Node->Op(DualListData.DataBegin());
|
||||
|
||||
if (IRSizes[Header->Op] >= sizeof(IROp_Constant)) {
|
||||
// Unlink any arguments the node currently has
|
||||
RemoveArgUses(Node);
|
||||
|
||||
// Overwrite data with the new constant op
|
||||
Header->Op = OP_CONSTANT;
|
||||
auto Const = Header->CW<IROp_Constant>();
|
||||
Const->Constant = Value;
|
||||
} else {
|
||||
// Fallback path for when the node to overwrite is too small
|
||||
auto cursor = GetWriteCursor();
|
||||
SetWriteCursor(Node);
|
||||
|
||||
auto NewNode = _Constant(Value);
|
||||
ReplaceAllUsesWith(Node, NewNode);
|
||||
|
||||
SetWriteCursor(cursor);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -182,11 +182,6 @@ public:
|
||||
return NodeIterator(DualListData.ListBegin(), DualListData.DataBegin(), wrapper);
|
||||
}
|
||||
|
||||
// Overwrite a node with a constant
|
||||
// Depending on what node has been overwritten, there might be some unallocated space around the node
|
||||
// Because we are overwriting the node, we don't have to worry about update all the arguments which use it
|
||||
void ReplaceWithConstant(Ref Node, uint64_t Value);
|
||||
|
||||
void ReplaceAllUsesWithRange(Ref Node, Ref NewNode, AllNodesIterator Begin, AllNodesIterator End);
|
||||
|
||||
void ReplaceUsesWithAfter(Ref Node, Ref NewNode, AllNodesIterator After) {
|
||||
@@ -201,24 +196,10 @@ public:
|
||||
ReplaceUsesWithAfter(Node, NewNode, It);
|
||||
}
|
||||
|
||||
void ReplaceAllUsesWith(Ref Node, Ref NewNode) {
|
||||
auto Start = AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin(), Node->Wrapped(DualListData.ListBegin()));
|
||||
|
||||
ReplaceAllUsesWithRange(Node, NewNode, Start, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
|
||||
LOGMAN_THROW_A_FMT(Node->NumUses == 0, "Node still used");
|
||||
|
||||
auto IROp = Node->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_Header>();
|
||||
// We can not remove the op if there are side-effects
|
||||
if (!IR::HasSideEffects(IROp->Op)) {
|
||||
// Since we have deleted ALL uses, we can safely delete the node.
|
||||
Remove(Node);
|
||||
}
|
||||
}
|
||||
|
||||
void ReplaceNodeArgument(Ref Node, uint8_t Arg, Ref NewArg);
|
||||
|
||||
void Remove(Ref Node);
|
||||
void RemovePostRA(Ref Node);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, Ref Src);
|
||||
Ref GetPackedRFLAG(bool Lower8);
|
||||
@@ -270,7 +251,8 @@ public:
|
||||
IRPair<IROp_CodeBlock> CreateCodeNode() {
|
||||
SetWriteCursor(nullptr); // Orphan from any previous nodes
|
||||
|
||||
auto CodeNode = _CodeBlock(InvalidNode, InvalidNode);
|
||||
auto ID = ViewIR().GetHeader()->BlockCount++;
|
||||
auto CodeNode = _CodeBlock(InvalidNode, InvalidNode, ID);
|
||||
|
||||
CodeBlocks.emplace_back(CodeNode);
|
||||
|
||||
|
||||
@@ -141,7 +141,7 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
Utils::FixedSizePooledAllocation<uintptr_t, 5000, 500> PoolObject;
|
||||
Utils::PoolBufferWithTimedRetirement<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
class IRListView final {
|
||||
@@ -234,6 +234,16 @@ public:
|
||||
return GetOp<IROp_IRHeader>(GetHeaderNode());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
unsigned PostRA() const {
|
||||
return GetHeader()->PostRA;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
unsigned SpillSlots() const {
|
||||
return GetHeader()->SpillSlots;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
[[nodiscard]]
|
||||
T* GetOp(Ref Node) const {
|
||||
@@ -409,9 +419,6 @@ class IRStorageBase {
|
||||
public:
|
||||
virtual ~IRStorageBase() = default;
|
||||
|
||||
// Optional RA data. Returns nullptr if none present
|
||||
virtual const RegisterAllocationData* RAData() = 0;
|
||||
|
||||
virtual IRListView GetIRView() = 0;
|
||||
};
|
||||
|
||||
|
||||
@@ -71,7 +71,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->GetGPROpSize()));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
}
|
||||
@@ -79,7 +79,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
void PassManager::AddDefaultValidationPasses() {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
InsertValidationPass(Validation::CreateIRValidation(), "IRValidation");
|
||||
InsertValidationPass(Validation::CreateRAValidation());
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class CPUIDEmu;
|
||||
struct HostFeatures;
|
||||
} // namespace FEXCore
|
||||
|
||||
@@ -15,16 +14,14 @@ class IntrusivePooledAllocator;
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateRAValidation();
|
||||
} // namespace Validation
|
||||
|
||||
namespace Debug {
|
||||
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
@@ -23,23 +22,6 @@ $end_info$
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size >= IR::OpSize::i8Bit && Op->Size <= IR::OpSize::i64Bit, "Invalid mask size");
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
// Returns true if the number bits from [0:width) contain the same bit.
|
||||
// Ensuring that the consecutive bits in the range are entirely 0 or 1.
|
||||
static bool HasConsecutiveBits(uint64_t imm, unsigned width) {
|
||||
if (width == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Credit to https://github.com/dougallj for this implementation.
|
||||
return ((imm ^ (imm >> 1)) & ((1ULL << (width - 1)) - 1)) == 0;
|
||||
}
|
||||
|
||||
// aarch64 heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
if (width < 32) {
|
||||
@@ -50,18 +32,15 @@ static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID)
|
||||
: SupportsTSOImm9 {SupportsTSOImm9}
|
||||
, CPUID {CPUID} {}
|
||||
explicit ConstProp(bool SupportsTSOImm9)
|
||||
: SupportsTSOImm9 {SupportsTSOImm9} {}
|
||||
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
void HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR);
|
||||
void ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
bool SupportsTSOImm9 {};
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
|
||||
template<class F>
|
||||
bool InlineIf(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index, F Filter) {
|
||||
@@ -122,100 +101,6 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
// Constants are pooled per block.
|
||||
void ConstProp::HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
const uint32_t SSACount = CurrentIR.GetSSACount();
|
||||
|
||||
// Allocation/initialization deferred until first use, since many multiblocks
|
||||
// don't have constants leftover after all inlining.
|
||||
fextl::vector<Ref> Remap {};
|
||||
|
||||
struct Entry {
|
||||
int64_t Value;
|
||||
Ref R;
|
||||
};
|
||||
|
||||
|
||||
fextl::vector<Entry> Pool {};
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
Pool.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
bool Found = false;
|
||||
|
||||
// Search for the constant. This is O(n^2) but n is small since it's
|
||||
// local and most constants are inlined. In practice, it ends up much
|
||||
// faster than a hash table.
|
||||
for (auto K : Pool) {
|
||||
if (K.Value == Op->Constant) {
|
||||
uint32_t Value = CurrentIR.GetID(CodeNode).Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "def not yet remapped");
|
||||
|
||||
if (Remap.empty()) {
|
||||
Remap.resize(SSACount, nullptr);
|
||||
}
|
||||
|
||||
Remap[Value] = K.R;
|
||||
Found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!Found) {
|
||||
Pool.push_back({.Value = Op->Constant, .R = CodeNode});
|
||||
}
|
||||
} else if (!Remap.empty()) {
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t Value = IROp->Args[i].ID().Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "src not yet remapped");
|
||||
|
||||
Ref New = Remap[Value];
|
||||
if (New) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, New);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Helper to replace the destination of an instruction with one of its sources,
|
||||
// to implement algebraic identities. This is surprisingly tricky due to
|
||||
// implicit masking in our IR.
|
||||
//
|
||||
// FEX's IR uses sized opcodes, matching arm64 semantics. 64-bit opcodes do not
|
||||
// mask, whereas smaller opcodes mask/zero-extend from 32-bits. Therefore, if
|
||||
// the instruction is 32-bit, we need to mask the source for a sound
|
||||
// replacement, in case there was garbage in the upper bits.
|
||||
//
|
||||
// However, if that source is in turn written by a 32-bit instruction, it is
|
||||
// guaranteed to have already been masked, so we know there's no garbage and we
|
||||
// can avoid the zero-extension. This is the case 99% of the time, but the
|
||||
// masking here is correctness-bearing nevertheless (and new versions of Denuvo
|
||||
// break if you get this wrong!)
|
||||
static inline void ReplaceWithSource(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Idx) {
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[Idx]);
|
||||
|
||||
if (IROp->Size < OpSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == OpSize::i32Bit, "other sizes not here");
|
||||
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[Idx]);
|
||||
if (Header->Size > OpSize::i32Bit) {
|
||||
Arg = IREmit->_Bfe(OpSize::i32Bit, 32, 0, Arg);
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
@@ -226,7 +111,6 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
bool IsConstant1 = IREmit->IsValueConstant(IROp->Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(IROp->Args[1], &Constant2);
|
||||
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
@@ -237,16 +121,6 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
|
||||
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
break;
|
||||
} else if (IsConstant1 && IsConstant2 && IROp->Op == OP_SUB) {
|
||||
uint64_t NewConstant = (Constant1 - Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
break;
|
||||
}
|
||||
|
||||
if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// So, negate the operation to negate (and inline) the constant.
|
||||
@@ -287,74 +161,10 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
uint64_t Constant1, Constant2;
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2) &&
|
||||
Op->Shift == IR::ShiftType::LSL) {
|
||||
// Optimize the LSL case when we know both sources are constant.
|
||||
// This is a pattern that shows up with direction flag calculations if DF was set just before the operation.
|
||||
uint64_t NewConstant = (Constant1 - (Constant2 << Op->ShiftAmount)) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_AND: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
bool Replaced = false;
|
||||
|
||||
// Order matter for short circuit evaluation, subsequent ifs read constant2.
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
uint64_t NewConstant = (Constant1 & Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Replaced = true;
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
Replaced = true;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_AND:
|
||||
case OP_OR:
|
||||
case OP_XOR: {
|
||||
uint64_t Constant1 {};
|
||||
|
||||
if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
// XOR with same value results to zero
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, IREmit->_Constant(0));
|
||||
} else {
|
||||
// XOR with zero results in the nonzero source
|
||||
bool Replaced = false;
|
||||
for (unsigned i = 0; i < 2; ++i) {
|
||||
if (!IREmit->IsValueConstant(IROp->Args[i], &Constant1)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Constant1 != 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 1 - i);
|
||||
Replaced = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
}
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_ANDWITHFLAGS:
|
||||
@@ -363,249 +173,19 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_NEG: {
|
||||
uint64_t Constant {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant)) {
|
||||
uint64_t NewConstant = -Constant;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ASHR:
|
||||
case OP_ROR: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHL: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == OpSize::i64Bit ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHR: {
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_BFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
|
||||
if (IROp->Size <= OpSize::i64Bit && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
uint64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
uint64_t DestMask = DestSizeInBits == 64 ? ~0ULL : ((1ULL << DestSizeInBits) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
NewConstant <<= 64 - Op->Width;
|
||||
NewConstant >>= 64 - Op->Width;
|
||||
NewConstant &= DestMask;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantSrc {};
|
||||
bool SrcIsConstant = IREmit->IsValueConstant(IROp->Args[1], &ConstantSrc);
|
||||
|
||||
if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) {
|
||||
// We are trying to insert constant, if it is a bitfield of only set bits then we can orr or and it.
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t NewConstant = SourceMask << Op->lsb;
|
||||
|
||||
if (ConstantSrc & 1) {
|
||||
auto orr = IREmit->_Or(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, orr);
|
||||
} else {
|
||||
// We are wanting to clear the bitfield.
|
||||
auto andn = IREmit->_Andn(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, andn);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (IROp->Size >= sourceHeader->Size &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)) {
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SYSCALL: {
|
||||
auto Op = IROp->CW<IR::IROp_Syscall>();
|
||||
|
||||
// Is the first argument a constant?
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->SyscallID, &Constant)) {
|
||||
auto SyscallDef = Manager->SyscallHandler->GetSyscallABI(Constant);
|
||||
auto SyscallFlags = Manager->SyscallHandler->GetSyscallFlags(Constant);
|
||||
|
||||
// Update the syscall flags
|
||||
Op->Flags = SyscallFlags;
|
||||
|
||||
// XXX: Once we have the ability to do real function calls then we can call directly in to the syscall handler
|
||||
if (SyscallDef.NumArgs < FEXCore::HLE::SyscallArguments::MAX_ARGS) {
|
||||
// If the number of args are less than what the IR op supports then we can remove arg usage
|
||||
// We need +1 since we are still passing in syscall number here
|
||||
for (uint8_t Arg = (SyscallDef.NumArgs + 1); Arg < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++Arg) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Arg, IREmit->Invalid());
|
||||
}
|
||||
// Replace syscall with inline passthrough syscall if we can
|
||||
if (SyscallDef.HostSyscallNumber != -1) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// Skip Args[0] since that is the syscallid
|
||||
auto InlineSyscall =
|
||||
IREmit->_InlineSyscall(CurrentIR.GetNode(IROp->Args[1]), CurrentIR.GetNode(IROp->Args[2]), CurrentIR.GetNode(IROp->Args[3]),
|
||||
CurrentIR.GetNode(IROp->Args[4]), CurrentIR.GetNode(IROp->Args[5]), CurrentIR.GetNode(IROp->Args[6]),
|
||||
SyscallDef.HostSyscallNumber, Op->Flags);
|
||||
|
||||
// Replace all syscall uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, InlineSyscall);
|
||||
|
||||
// We must remove here since DCE can't remove a IROp with sideeffects
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CPUID: {
|
||||
auto Op = IROp->CW<IR::IROp_CPUID>();
|
||||
|
||||
uint64_t ConstantFunction {}, ConstantLeaf {};
|
||||
bool IsConstantFunction = IREmit->IsValueConstant(Op->Function, &ConstantFunction);
|
||||
bool IsConstantLeaf = IREmit->IsValueConstant(Op->Leaf, &ConstantLeaf);
|
||||
// If the CPUID function is constant then we can try and optimize.
|
||||
if (IsConstantFunction) { // && ConstantFunction != 1) {
|
||||
// Check if it supports constant data reporting for this function.
|
||||
const auto SupportsConstant = CPUID->DoesFunctionReportConstantData(ConstantFunction);
|
||||
if (SupportsConstant.SupportsConstantFunction == CPUIDEmu::SupportsConstant::CONSTANT) {
|
||||
// If the CPUID needs a constant leaf to be optimized then this can't work if we didn't const-prop the leaf register.
|
||||
if (!(SupportsConstant.NeedsLeaf == CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT && !IsConstantLeaf)) {
|
||||
// Calculate the constant data and replace all uses.
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEBX), IREmit->_Constant(Result.ebx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutECX), IREmit->_Constant(Result.ecx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_XGETBV: {
|
||||
auto Op = IROp->CW<IR::IROp_XGetBV>();
|
||||
|
||||
uint64_t ConstantFunction {};
|
||||
if (IREmit->IsValueConstant(Op->Function, &ConstantFunction) && CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LDIV:
|
||||
case OP_LREM: {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
auto UpperIROp = IREmit->GetOpHeader(Op->Upper);
|
||||
|
||||
// Check upper Op to see if it came from a sign-extension
|
||||
if (UpperIROp->Op != OP_SBFE) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Sbfe = UpperIROp->C<IR::IROp_Sbfe>();
|
||||
if (Sbfe->Width != 1 || Sbfe->lsb != 63 || Sbfe->Header.Args[0] != Op->Lower) {
|
||||
break;
|
||||
}
|
||||
|
||||
// If it does then it we only need a 64bit SDIV
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Lower = CurrentIR.GetNode(Op->Lower);
|
||||
Ref Divisor = CurrentIR.GetNode(Op->Divisor);
|
||||
Ref SDivOp {};
|
||||
if (IROp->Op == OP_LDIV) {
|
||||
SDivOp = IREmit->_Div(OpSize::i64Bit, Lower, Divisor);
|
||||
} else {
|
||||
SDivOp = IREmit->_Rem(OpSize::i64Bit, Lower, Divisor);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, SDivOp);
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LUDIV:
|
||||
case OP_LUREM: {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
// Check upper Op to see if it came from a zeroing op
|
||||
// If it does then it we only need a 64bit UDIV
|
||||
uint64_t Value;
|
||||
if (!IREmit->IsValueConstant(Op->Upper, &Value) || Value != 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Lower = CurrentIR.GetNode(Op->Lower);
|
||||
Ref Divisor = CurrentIR.GetNode(Op->Divisor);
|
||||
Ref UDivOp {};
|
||||
if (IROp->Op == OP_LUDIV) {
|
||||
UDivOp = IREmit->_UDiv(OpSize::i64Bit, Lower, Divisor);
|
||||
} else {
|
||||
UDivOp = IREmit->_URem(OpSize::i64Bit, Lower, Divisor);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, UDivOp);
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_RMIFNZCV: {
|
||||
@@ -747,15 +327,75 @@ void ConstProp::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
const uint32_t SSACount = CurrentIR.GetSSACount();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
// Allocation/initialization deferred until first use, since many multiblocks
|
||||
// don't have constants leftover after all inlining.
|
||||
fextl::vector<Ref> Remap {};
|
||||
|
||||
struct Entry {
|
||||
int64_t Value;
|
||||
Ref R;
|
||||
};
|
||||
|
||||
fextl::vector<Entry> Pool {};
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
Pool.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
bool Found = false;
|
||||
|
||||
// Search for the constant. This is O(n^2) but n is small since it's
|
||||
// local and most constants are inlined. In practice, it ends up much
|
||||
// faster than a hash table.
|
||||
for (auto K : Pool) {
|
||||
if (K.Value == Op->Constant) {
|
||||
uint32_t Value = CurrentIR.GetID(CodeNode).Value;
|
||||
if (Value < SSACount) {
|
||||
if (Remap.empty()) {
|
||||
Remap.resize(SSACount, nullptr);
|
||||
}
|
||||
|
||||
Remap[Value] = K.R;
|
||||
}
|
||||
Found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!Found) {
|
||||
Pool.push_back({.Value = Op->Constant, .R = CodeNode});
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
|
||||
if (!Remap.empty()) {
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t Value = IROp->Args[i].ID().Value;
|
||||
if (Value < SSACount) {
|
||||
Ref New = Remap[Value];
|
||||
if (New) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, New);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
HandleConstantPools(IREmit, IREmit->ViewIR());
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID) {
|
||||
return fextl::make_unique<ConstProp>(SupportsTSOImm9, CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9) {
|
||||
return fextl::make_unique<ConstProp>(SupportsTSOImm9);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -38,12 +38,6 @@ IRDumper::IRDumper() {
|
||||
}
|
||||
|
||||
void IRDumper::Run(IREmitter* IREmit) {
|
||||
auto RAPass = Manager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
IR::RegisterAllocationData* RA {};
|
||||
if (RAPass) {
|
||||
RA = RAPass->GetAllocationData();
|
||||
}
|
||||
|
||||
FEXCore::File::File FD {};
|
||||
if (DumpIR() == "stderr") {
|
||||
FD = FEXCore::File::File::GetStdERR();
|
||||
@@ -57,18 +51,18 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpToFile) {
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, IR.PostRA() ? "-post.ir" : "-pre.ir");
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE | FEXCore::File::FileModes::CREATE | FEXCore::File::FileModes::TRUNCATE);
|
||||
}
|
||||
|
||||
if (FD.IsValid() || DumpToLog) {
|
||||
fextl::stringstream out;
|
||||
FEXCore::IR::Dump(&out, &IR, RA);
|
||||
FEXCore::IR::Dump(&out, &IR);
|
||||
if (FD.IsValid()) {
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
} else {
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -58,11 +58,6 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
#endif
|
||||
|
||||
IR::RegisterAllocationData* RAData {};
|
||||
if (Manager->HasPass("RA")) {
|
||||
RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
}
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
@@ -94,9 +89,9 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
if (RAData) {
|
||||
// If we have a register allocator then the destination needs to be assigned a register and class
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
if (CurrentIR.PostRA()) {
|
||||
// After RA, the destination needs to be assigned a register and class
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
|
||||
FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType {PhyReg.Class};
|
||||
@@ -127,6 +122,10 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
OrderedNodeWrapper Arg = IROp->Args[i];
|
||||
const auto ArgID = Arg.ID();
|
||||
if (Arg.IsImmediate()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
@@ -239,18 +238,21 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
// Use counts are only relevant pre-RA.
|
||||
if (!CurrentIR.PostRA()) {
|
||||
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
HadWarning = false;
|
||||
if (HadError || HadWarning) {
|
||||
fextl::stringstream Out;
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR, RAData);
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR);
|
||||
|
||||
if (HadError) {
|
||||
Out << "Errors:" << std::endl << Errors.str() << std::endl;
|
||||
|
||||
@@ -16,8 +16,6 @@ struct BlockInfo {
|
||||
fextl::vector<OrderedNode*> Successors;
|
||||
};
|
||||
|
||||
class RAValidation;
|
||||
|
||||
class IRValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~IRValidation();
|
||||
@@ -29,7 +27,5 @@ private:
|
||||
OrderedNode* EntryBlock {};
|
||||
fextl::unordered_map<IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
size_t MaxNodes {};
|
||||
|
||||
friend class RAValidation;
|
||||
};
|
||||
} // namespace FEXCore::IR::Validation
|
||||
@@ -1,197 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Interface/IR/Passes/IRValidation.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace FEXCore::IR::Validation {
|
||||
|
||||
// Hold the mapping of physical registers to the SSA id it holds at any given point in the IR
|
||||
struct RegState {
|
||||
static constexpr IR::NodeID UninitializedValue {0};
|
||||
static constexpr IR::NodeID InvalidReg {0xffff'ffff};
|
||||
|
||||
// This class makes some assumptions about how the host registers are arranged and mapped to virtual registers:
|
||||
// 1. There will be less than 32 GPRs and 32 FPRs
|
||||
// 2. If the GPRFixed class is used, there will be 16 GPRs and 16 FixedGPRs max
|
||||
// 3. Same with FPRFixed
|
||||
|
||||
// These assumptions were all true for the state of the arm64 and x86 jits at the time this was written
|
||||
|
||||
// Mark a physical register as containing a SSA id
|
||||
bool Set(PhysicalRegister Reg, IR::NodeID ssa) {
|
||||
LOGMAN_THROW_A_FMT(ssa.IsValid(), "RegState assumes ssa0 will be the block header and never assigned to a register");
|
||||
|
||||
// PhysicalRegisters aren't fully mapped until assembly emission
|
||||
// We need to apply a generic mapping here to catch any aliasing
|
||||
switch (Reg.Class) {
|
||||
case GPRClass: GPRs[Reg.Reg] = ssa; return true;
|
||||
case GPRFixedClass:
|
||||
// On arm64, there are 16 Fixed and 9 normal
|
||||
GPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case FPRClass: FPRs[Reg.Reg] = ssa; return true;
|
||||
case FPRFixedClass:
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Get the current SSA id
|
||||
// Or an error value there isn't a (sane) SSA id
|
||||
IR::NodeID Get(PhysicalRegister Reg) const {
|
||||
switch (Reg.Class) {
|
||||
case GPRClass: return GPRs[Reg.Reg];
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
|
||||
// Mark a spill slot as containing a SSA id
|
||||
void Spill(uint32_t SpillSlot, IR::NodeID ssa) {
|
||||
Spills[SpillSlot] = ssa;
|
||||
}
|
||||
|
||||
// Return the SSA id currently in a spill slot
|
||||
IR::NodeID Unspill(uint32_t SpillSlot) {
|
||||
if (Spills.contains(SpillSlot)) {
|
||||
return Spills[SpillSlot];
|
||||
} else {
|
||||
return UninitializedValue;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
std::array<IR::NodeID, 32> GPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> FPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> GPRs = {};
|
||||
std::array<IR::NodeID, 32> FPRs = {};
|
||||
|
||||
fextl::unordered_map<uint32_t, IR::NodeID> Spills;
|
||||
};
|
||||
|
||||
class RAValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~RAValidation() {}
|
||||
void Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
|
||||
void RAValidation::Run(IREmitter* IREmit) {
|
||||
if (!Manager->HasPass("RA")) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
|
||||
bool HadError = false;
|
||||
fextl::ostringstream Errors;
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
// We only allocate registers locally, so state is reset each block
|
||||
struct RegState BlockRegState = {};
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto ID = CurrentIR.GetID(CodeNode);
|
||||
|
||||
const auto CheckArg = [&](uint32_t i, OrderedNodeWrapper Arg) {
|
||||
const auto ArgID = Arg.ID();
|
||||
const auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
if (PhyReg.IsInvalid()) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto CurrentSSAAtReg = BlockRegState.Get(PhyReg);
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n", ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg != ArgID) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n", ID, i, PhyReg.Reg,
|
||||
ArgID, CurrentSSAAtReg);
|
||||
}
|
||||
};
|
||||
|
||||
switch (IROp->Op) {
|
||||
case OP_SPILLREGISTER: {
|
||||
auto SpillRegister = IROp->C<IROp_SpillRegister>();
|
||||
CheckArg(0, SpillRegister->Value);
|
||||
|
||||
BlockRegState.Spill(SpillRegister->Slot, SpillRegister->Value.ID());
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_FILLREGISTER: {
|
||||
auto FillRegister = IROp->C<IROp_FillRegister>();
|
||||
const auto ExpectedValue = FillRegister->OriginalValue.ID();
|
||||
const auto Value = BlockRegState.Unspill(FillRegister->Slot);
|
||||
|
||||
// TODO: This only proves that the Spill has a consistent SSA value
|
||||
// In the future we need to prove it contains the correct SSA value. For
|
||||
// this we need to analyze copies/swaps properly. As a hot fix, don't
|
||||
// compare Value with ExpectedValue.
|
||||
|
||||
if (Value == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined\n", ID, ExpectedValue, FillRegister->Slot);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: {
|
||||
// And check that all args point at the correct SSA
|
||||
uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
CheckArg(i, IROp->Args[i]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Update BlockState map
|
||||
if (IROp->Op != OP_SPILLREGISTER) {
|
||||
BlockRegState.Set(RAData->GetNodeRegister(ID), ID);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (HadError) {
|
||||
fextl::stringstream IrDump;
|
||||
FEXCore::IR::Dump(&IrDump, &CurrentIR, RAData);
|
||||
|
||||
LogMan::Msg::EFmt("RA Validation Error\n{}\nErrors:\n{}\n", IrDump.str(), Errors.str());
|
||||
LOGMAN_MSG_A_FMT("Encountered RA validation Error");
|
||||
|
||||
Errors.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateRAValidation() {
|
||||
return fextl::make_unique<RAValidation>();
|
||||
}
|
||||
} // namespace FEXCore::IR::Validation
|
||||
@@ -97,46 +97,49 @@ private:
|
||||
};
|
||||
|
||||
struct BlockInfo {
|
||||
fextl::vector<Ref> Predecessors;
|
||||
fextl::vector<uint32_t> Predecessors;
|
||||
Ref Node;
|
||||
uint8_t Flags;
|
||||
bool InWorklist;
|
||||
};
|
||||
|
||||
struct ControlFlowGraph {
|
||||
fextl::unordered_map<uint32_t, BlockInfo> BlockMap;
|
||||
fextl::vector<BlockInfo> BlockMap;
|
||||
IRListView& IR;
|
||||
|
||||
void AddBlock(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
uint32_t ID = IR.GetID(Block).Value;
|
||||
void Init(fextl::deque<uint32_t>& Worklist, uint32_t BlockCount) {
|
||||
BlockMap.resize(BlockCount);
|
||||
|
||||
// Add the block with conservative flags and already in the worklist.
|
||||
auto Info = &BlockMap.emplace(ID, BlockInfo {{}, FLAG_ALL, true}).first->second;
|
||||
for (unsigned ID = 0; ID < BlockCount; ++ID) {
|
||||
// Add the block with conservative flags and already in the worklist.
|
||||
auto Info = BlockInfo {{}, nullptr, FLAG_ALL, true};
|
||||
|
||||
// Add some initial capacity
|
||||
Info->Predecessors.reserve(2);
|
||||
// Add some initial capacity
|
||||
Info.Predecessors.reserve(2);
|
||||
|
||||
// Add to worklist
|
||||
Worklist.push_back(Block);
|
||||
BlockMap[ID] = Info;
|
||||
Worklist.push_back(ID);
|
||||
}
|
||||
}
|
||||
|
||||
BlockInfo* Get(uint32_t Block) {
|
||||
return &BlockMap.try_emplace(Block).first->second;
|
||||
return &BlockMap[Block];
|
||||
}
|
||||
|
||||
BlockInfo* Get(Ref Block) {
|
||||
return Get(IR.GetID(Block).Value);
|
||||
BlockInfo* Get(IROp_CodeBlock* Block) {
|
||||
return &BlockMap[Block->ID];
|
||||
}
|
||||
|
||||
BlockInfo* Get(OrderedNodeWrapper Block) {
|
||||
return Get(Block.ID().Value);
|
||||
return Get(IR.GetOp<IR::IROp_CodeBlock>(Block));
|
||||
}
|
||||
|
||||
void RecordEdge(Ref From, Ref To) {
|
||||
void RecordEdge(uint32_t From, OrderedNodeWrapper To) {
|
||||
auto Info = Get(To);
|
||||
Info->Predecessors.push_back(From);
|
||||
}
|
||||
|
||||
void AddWorklist(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
void AddWorklist(fextl::deque<uint32_t>& Worklist, uint32_t Block) {
|
||||
auto Info = Get(Block);
|
||||
if (!Info->InWorklist) {
|
||||
Info->InWorklist = true;
|
||||
@@ -637,8 +640,8 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie
|
||||
// For the purposes of global propagation, the content of our progress doesn't
|
||||
// matter -- only the difference in our final FlagsRead contributes to changes
|
||||
// in the predecessors.
|
||||
uint32_t OldFlagsRead = CFG.Get(Block)->Flags;
|
||||
CFG.Get(Block)->Flags = FlagsRead;
|
||||
uint32_t OldFlagsRead = CFG.Get(BlockIROp->ID)->Flags;
|
||||
CFG.Get(BlockIROp->ID)->Flags = FlagsRead;
|
||||
return (OldFlagsRead != FlagsRead);
|
||||
}
|
||||
|
||||
@@ -650,12 +653,14 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
// Initialize conservatively: all blocks need full parity. This initialization
|
||||
// matters for proper handling of backedges.
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
CFG.Get(Block)->Flags = FULL;
|
||||
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
CFG.Get(ID)->Flags = FULL;
|
||||
}
|
||||
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
bool Full = false;
|
||||
auto Predecessors = CFG.Get(Block)->Predecessors;
|
||||
auto Predecessors = CFG.Get(ID)->Predecessors;
|
||||
|
||||
if (Predecessors.empty()) {
|
||||
// Conservatively assume there was full parity before the start block
|
||||
@@ -701,7 +706,7 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
}
|
||||
|
||||
// Record our final state for our successors to read.
|
||||
CFG.Get(Block)->Flags = Full ? FULL : PARTIAL;
|
||||
CFG.Get(ID)->Flags = Full ? FULL : PARTIAL;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -709,28 +714,28 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::deque<Ref> Worklist;
|
||||
fextl::deque<uint32_t> Worklist;
|
||||
|
||||
// Initialize CFG
|
||||
ControlFlowGraph CFG {.IR = CurrentIR};
|
||||
|
||||
// Gather blocks
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
CFG.AddBlock(Worklist, BlockNode);
|
||||
}
|
||||
CFG.Init(Worklist, CurrentIR.GetHeader()->BlockCount);
|
||||
|
||||
// Gather CFG
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto CodeLast = CurrentIR.at(BlockHeader->C<IROp_CodeBlock>()->Last);
|
||||
auto Block = BlockHeader->C<IROp_CodeBlock>();
|
||||
auto CodeLast = CurrentIR.at(Block->Last);
|
||||
--CodeLast;
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->TrueBlock));
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->FalseBlock));
|
||||
CFG.RecordEdge(Block->ID, Op->TrueBlock);
|
||||
CFG.RecordEdge(Block->ID, Op->FalseBlock);
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(ExitOp->Args[0]));
|
||||
CFG.RecordEdge(Block->ID, ExitOp->Args[0]);
|
||||
}
|
||||
|
||||
CFG.Get(Block->ID)->Node = BlockNode;
|
||||
}
|
||||
|
||||
// After processing a block, if we made progress, we must process its
|
||||
@@ -741,7 +746,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
auto Info = CFG.Get(Block);
|
||||
Info->InWorklist = false;
|
||||
|
||||
if (ProcessBlock(IREmit, CurrentIR, Block, CFG)) {
|
||||
if (ProcessBlock(IREmit, CurrentIR, Info->Node, CFG)) {
|
||||
for (auto Pred : Info->Predecessors) {
|
||||
CFG.AddWorklist(Worklist, Pred);
|
||||
}
|
||||
|
||||
@@ -28,9 +28,9 @@ namespace {
|
||||
uint32_t Available;
|
||||
uint32_t Count;
|
||||
|
||||
// If bit R of Available is 0, then RegToSSA[R] is the Old node
|
||||
// currently allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to
|
||||
// clear this when freeing registers.
|
||||
// If bit R of Available is 0, then RegToSSA[R] is the node currently
|
||||
// allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to clear this
|
||||
// when freeing registers.
|
||||
Ref RegToSSA[32];
|
||||
};
|
||||
|
||||
@@ -57,100 +57,29 @@ class ConstrainedRAPass final : public RegisterAllocationPass {
|
||||
public:
|
||||
void Run(IREmitter* IREmit) override;
|
||||
void AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) override;
|
||||
|
||||
RegisterAllocationData* GetAllocationData() override;
|
||||
RegisterAllocationData::UniquePtr PullAllocationData() override;
|
||||
bool TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
private:
|
||||
IR::RegisterAllocationData::UniquePtr AllocData;
|
||||
RegisterClass Classes[INVALID_CLASS];
|
||||
|
||||
IREmitter* IREmit;
|
||||
IRListView* IR;
|
||||
|
||||
// Map of Old nodes to their preferred register, to coalesce load/store reg.
|
||||
// Map of nodes to their preferred register, to coalesce load/store reg.
|
||||
fextl::vector<PhysicalRegister> PreferredReg;
|
||||
|
||||
// FEX's original RA could only assign a single register to a given def for
|
||||
// its entire live range, and this limitation is baked deep into the IR.
|
||||
// However, we split live ranges to implement register pairs and spilling.
|
||||
//
|
||||
// To reconcile, we generate new SSA nodes when we split live ranges, and
|
||||
// remap SSA sources accordingly. This means SSAToReg can grow.
|
||||
//
|
||||
// We define "Old" nodes as nodes present in the original IR, and "New" nodes
|
||||
// as nodes added to split live ranges. Helpful properties:
|
||||
//
|
||||
// - A node is Old <===> it is not New
|
||||
// - A node is Old <===> its ID < IR.GetSSACount() at the start
|
||||
// - All sources are Old before remapping an instruction
|
||||
//
|
||||
// SSAToNewSSA tracks the current remapping. nullptr indicates no remapping.
|
||||
//
|
||||
// Since its indexed by Old nodes, SSAToNewSSA does not grow after allocation.
|
||||
fextl::vector<Ref> SSAToNewSSA;
|
||||
|
||||
// Inverse of SSAToNewSSA. Since it's indexed by new nodes, it grows.
|
||||
fextl::vector<Ref> NewSSAToSSA;
|
||||
|
||||
// Map of assigned registers. Grows.
|
||||
// Map of assigned registers. Does not grow beyond the initial set.
|
||||
fextl::vector<PhysicalRegister> SSAToReg;
|
||||
|
||||
bool IsOld(Ref Node) {
|
||||
return IR->GetID(Node).Value < PreferredReg.size();
|
||||
};
|
||||
|
||||
// Return the New node (if it exists) for an Old node, else the Old node.
|
||||
Ref Map(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Pre-condition");
|
||||
|
||||
if (SSAToNewSSA.empty()) {
|
||||
return Old;
|
||||
} else {
|
||||
return SSAToNewSSA[IR->GetID(Old).Value] ?: Old;
|
||||
}
|
||||
};
|
||||
|
||||
// Return the Old node for a possibly-remapped node.
|
||||
Ref Unmap(Ref Node) {
|
||||
if (NewSSAToSSA.empty()) {
|
||||
return Node;
|
||||
} else {
|
||||
return NewSSAToSSA[IR->GetID(Node).Value] ?: Node;
|
||||
}
|
||||
};
|
||||
|
||||
// Record a remapping of Old to New.
|
||||
void Remap(Ref Old, Ref New) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old) && !IsOld(New), "Pre-condition");
|
||||
|
||||
uint32_t OldID = IR->GetID(Old).Value;
|
||||
uint32_t NewID = IR->GetID(New).Value;
|
||||
|
||||
LOGMAN_THROW_A_FMT(NewID >= NewSSAToSSA.size(), "Brand new SSA def");
|
||||
NewSSAToSSA.resize(NewID + 1, 0);
|
||||
|
||||
if (SSAToNewSSA.empty()) {
|
||||
SSAToNewSSA.resize(PreferredReg.size(), nullptr);
|
||||
}
|
||||
|
||||
SSAToNewSSA[OldID] = New;
|
||||
NewSSAToSSA[NewID] = Old;
|
||||
|
||||
LOGMAN_THROW_A_FMT(Map(Old) == New && Unmap(New) == Old, "Post-condition");
|
||||
LOGMAN_THROW_A_FMT(Unmap(Old) == Old, "Invariant1");
|
||||
};
|
||||
|
||||
// Maps Old defs to their assigned spill slot + 1, or 0 if not spilled.
|
||||
// Maps defs to their assigned spill slot + 1, or 0 if not spilled.
|
||||
fextl::vector<unsigned> SpillSlots;
|
||||
|
||||
bool Rematerializable(IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT;
|
||||
}
|
||||
|
||||
Ref InsertFill(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition");
|
||||
IROp_Header* IROp = IR->GetOp<IROp_Header>(Old);
|
||||
Ref InsertFill(Ref Node) {
|
||||
IROp_Header* IROp = IR->GetOp<IROp_Header>(Node);
|
||||
|
||||
// Remat if we can
|
||||
if (Rematerializable(IROp)) {
|
||||
@@ -159,22 +88,17 @@ private:
|
||||
}
|
||||
|
||||
// Otherwise fill from stack
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Old).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Old must have been spilled");
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
|
||||
|
||||
RegisterClassType RegClass = GetRegClassFromNode(IR, IROp);
|
||||
|
||||
auto Fill = IREmit->_FillRegister(Old, SlotPlusOne - 1, RegClass);
|
||||
Fill.first->Header.Size = IROp->Size;
|
||||
Fill.first->Header.ElementSize = IROp->ElementSize;
|
||||
return Fill;
|
||||
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
|
||||
};
|
||||
|
||||
// IP of next-use of each Old source. IPs are measured from the end of the
|
||||
// IP of next-use of each source. IPs are measured from the end of the
|
||||
// block, so we don't need to size the block up-front.
|
||||
fextl::vector<uint32_t> NextUses;
|
||||
|
||||
unsigned SpillSlotCount;
|
||||
bool AnySpilled;
|
||||
|
||||
bool IsValidArg(OrderedNodeWrapper Arg) {
|
||||
@@ -201,13 +125,14 @@ private:
|
||||
return 1 << Reg.Reg;
|
||||
};
|
||||
|
||||
bool IsInRegisterFile(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition");
|
||||
bool IsInRegisterFile(Ref Node) {
|
||||
auto ID = IR->GetID(Node).Value;
|
||||
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
PhysicalRegister Reg = SSAToReg[ID];
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Old;
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
|
||||
};
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
@@ -219,14 +144,9 @@ private:
|
||||
Class->Available |= RegBits;
|
||||
};
|
||||
|
||||
bool HasSource(IROp_Header* I, Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Invariant2");
|
||||
|
||||
bool HasSource(IROp_Header* I, PhysicalRegister Reg) {
|
||||
for (auto s = 0; s < IR::GetRAArgs(I->Op); ++s) {
|
||||
Ref Node = IR->GetNode(I->Args[s]);
|
||||
LOGMAN_THROW_A_FMT(IsOld(Node), "not yet mapped");
|
||||
|
||||
if (Node == Old) {
|
||||
if (I->Args[s].IsImmediate() && PhysicalRegister(I->Args[s]) == Reg) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -278,6 +198,18 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
bool IsTrivial(Ref Node, const IROp_Header* Header) {
|
||||
switch (Header->Op) {
|
||||
case OP_ALLOCATEGPR: return true;
|
||||
case OP_ALLOCATEGPRAFTER: return true;
|
||||
case OP_ALLOCATEFPR: return true;
|
||||
case OP_RMWHANDLE: return PhysicalRegister(Node) == PhysicalRegister(Header->Args[0]);
|
||||
case OP_LOADREGISTER: return PhysicalRegister(Node) == DecodeSRAReg(Header);
|
||||
case OP_STOREREGISTER: return PhysicalRegister(Header->Args[0]) == DecodeSRAReg(Header);
|
||||
default: return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Helper macro to walk the set bits b in a 32-bit word x, using ffs to get
|
||||
// the next set bit and then clearing on each iteration.
|
||||
#define foreach_bit(b, x) for (uint32_t __x = (x), b; ((b) = __builtin_ffs(__x) - 1, __x); __x &= ~(1 << (b)))
|
||||
@@ -292,33 +224,33 @@ private:
|
||||
uint32_t Allocated = ((1u << Class->Count) - 1) & ~Class->Available;
|
||||
|
||||
foreach_bit(i, Allocated) {
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
Ref Node = Class->RegToSSA[i];
|
||||
auto Reg = SSAToReg[IR->GetID(Node).Value];
|
||||
|
||||
LOGMAN_THROW_A_FMT(Old != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(SSAToReg[IR->GetID(Map(Old)).Value].Reg == i, "Invariant4");
|
||||
LOGMAN_THROW_A_FMT(Node != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(Reg.Reg == i, "Invariant4");
|
||||
|
||||
// Skip any source used by the current instruction, it is unspillable.
|
||||
if (!HasSource(Exclude, Old)) {
|
||||
uint32_t NextUse = NextUses[IR->GetID(Old).Value];
|
||||
if (!HasSource(Exclude, Reg)) {
|
||||
uint32_t NextUse = NextUses[IR->GetID(Node).Value];
|
||||
|
||||
// Prioritize remat over spilling. It is typically cheaper to remat a
|
||||
// constant multiple times than to spill a single value.
|
||||
if (!Rematerializable(IR->GetOp<IROp_Header>(Old))) {
|
||||
if (!Rematerializable(IR->GetOp<IROp_Header>(Node))) {
|
||||
NextUse += 100000;
|
||||
}
|
||||
|
||||
if (NextUse < BestDistance) {
|
||||
BestDistance = NextUse;
|
||||
BestReg = i;
|
||||
Candidate = Old;
|
||||
Candidate = Node;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Candidate != nullptr, "must've found something..");
|
||||
LOGMAN_THROW_A_FMT(IsOld(Candidate), "Invariant5");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Candidate)).Value];
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Candidate).Value];
|
||||
LOGMAN_THROW_A_FMT(Reg.Reg == BestReg, "Invariant6");
|
||||
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Candidate);
|
||||
@@ -336,10 +268,10 @@ private:
|
||||
}
|
||||
|
||||
// TODO: we should colour spill slots
|
||||
uint32_t Slot = SpillSlotCount++;
|
||||
uint32_t Slot = IR->GetHeader()->SpillSlots++;
|
||||
|
||||
// We must map here in case we're spilling something we shuffled.
|
||||
auto SpillOp = IREmit->_SpillRegister(Map(Candidate), Slot, RegisterClassType {Reg.Class});
|
||||
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, RegisterClassType {Reg.Class});
|
||||
SpillOp.first->Header.Size = Header->Size;
|
||||
SpillOp.first->Header.ElementSize = Header->ElementSize;
|
||||
SpillSlots[Value] = Slot + 1;
|
||||
@@ -350,22 +282,27 @@ private:
|
||||
AnySpilled = true;
|
||||
};
|
||||
|
||||
void RemapReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
Class->RegToSSA[Reg.Reg] = Node;
|
||||
|
||||
uint32_t Index = IR->GetID(Node).Value;
|
||||
if (Index < SSAToReg.size()) {
|
||||
SSAToReg[Index] = Reg;
|
||||
}
|
||||
};
|
||||
|
||||
// Record a given assignment of register Reg to Node.
|
||||
void SetReg(Ref Node, PhysicalRegister Reg) {
|
||||
uint32_t Index = IR->GetID(Node).Value;
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
|
||||
Class->Available &= ~RegBits;
|
||||
Class->RegToSSA[Reg.Reg] = Unmap(Node);
|
||||
|
||||
if (Index >= SSAToReg.size()) {
|
||||
SSAToReg.resize(Index + 1, PhysicalRegister::Invalid());
|
||||
}
|
||||
|
||||
SSAToReg[Index] = Reg;
|
||||
RemapReg(Node, Reg);
|
||||
Node->Reg = Reg.Raw;
|
||||
};
|
||||
|
||||
// Assign a register for a given Node, spilling if necessary.
|
||||
@@ -387,7 +324,7 @@ private:
|
||||
|
||||
// Try to handle tied registers. This can fail, the JIT will insert moves.
|
||||
if (int TiedIdx = IR::TiedSource(IROp->Op); TiedIdx >= 0) {
|
||||
PhysicalRegister Reg = SSAToReg[IROp->Args[TiedIdx].ID().Value];
|
||||
auto Reg = PhysicalRegister(IROp->Args[TiedIdx]);
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
@@ -417,7 +354,7 @@ private:
|
||||
}
|
||||
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
auto After = SSAToReg[IR->GetID(IR->GetNode(IROp->Args[0])).Value];
|
||||
auto After = PhysicalRegister(IROp->Args[0]);
|
||||
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
|
||||
return;
|
||||
@@ -438,10 +375,6 @@ private:
|
||||
unsigned Reg = std::countr_zero(Class->Available);
|
||||
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
|
||||
};
|
||||
|
||||
bool IsRAOp(IROps Op) {
|
||||
return Op == OP_SPILLREGISTER || Op == OP_FILLREGISTER || Op == OP_COPY;
|
||||
};
|
||||
};
|
||||
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) {
|
||||
@@ -450,12 +383,35 @@ void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t Regis
|
||||
Classes[Class].Count = RegisterCount;
|
||||
}
|
||||
|
||||
RegisterAllocationData* ConstrainedRAPass::GetAllocationData() {
|
||||
return AllocData.get();
|
||||
}
|
||||
bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp) {
|
||||
if (IROp->Op == OP_PUSH) {
|
||||
auto LastOp = IR->GetOp<IROp_Header>(LastNode);
|
||||
if (LastOp->Op == OP_PUSH) {
|
||||
auto SP = PhysicalRegister(CodeNode);
|
||||
auto Push = IR->GetOp<IROp_Push>(CodeNode);
|
||||
auto LastPush = IR->GetOp<IROp_Push>(LastNode);
|
||||
|
||||
RegisterAllocationData::UniquePtr ConstrainedRAPass::PullAllocationData() {
|
||||
return std::move(AllocData);
|
||||
if (LastOp->Size == IROp->Size && LastPush->ValueSize == Push->ValueSize && SP == PhysicalRegister(LastNode) &&
|
||||
SP == PhysicalRegister(IROp->Args[1]) && SP == PhysicalRegister(LastOp->Args[1]) && SP != PhysicalRegister(IROp->Args[0]) &&
|
||||
SP != PhysicalRegister(LastOp->Args[0]) && Push->ValueSize >= OpSize::i32Bit) {
|
||||
|
||||
IREmit->SetWriteCursorBefore(LastNode);
|
||||
IREmit->_PushTwo(IROp->Size, Push->ValueSize, IROp->Args[0], LastOp->Args[0], IROp->Args[1]);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
} else if (IROp->Op == OP_POP) {
|
||||
auto LastOp = IR->GetOp<IROp_Header>(LastNode);
|
||||
auto SP = PhysicalRegister(IROp->Args[0]);
|
||||
|
||||
if (LastOp->Op == OP_POP && LastOp->Size == IROp->Size && IROp->Size >= OpSize::i32Bit && SP == PhysicalRegister(LastOp->Args[0])) {
|
||||
IREmit->SetWriteCursorBefore(LastNode);
|
||||
IREmit->_PopTwo(IROp->Size, IROp->Args[0], LastOp->Args[1], IROp->Args[1]);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
@@ -465,11 +421,9 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
auto IR_ = IREmit->ViewIR();
|
||||
IR = &IR_;
|
||||
|
||||
// SSAToNewSSA, NewSSAToSSA allocated on first-use
|
||||
PreferredReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
SSAToReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
NextUses.resize(IR->GetSSACount(), 0);
|
||||
SpillSlotCount = 0;
|
||||
AnySpilled = false;
|
||||
|
||||
// Next-use distance relative to the block end of each source, last first.
|
||||
@@ -560,9 +514,17 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
int64_t SourceIndex = SourcesNextUses.size();
|
||||
|
||||
// Forward pass: Assign registers, spilling as we go.
|
||||
// Last nontrivial instruction, for merging as we go.
|
||||
Ref LastNode = nullptr;
|
||||
|
||||
// Forward pass: Assign registers, spilling & optimizing as we go.
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
LOGMAN_THROW_A_FMT(!IsRAOp(IROp->Op), "RA ops inserted before, so not seen iterating forward");
|
||||
// GuestOpcode does not read or write registers, and must be skipped for
|
||||
// push/pop merging. Since we'd be doing this check anyway for merging, do
|
||||
// the check now so we can skip the rest of the logic too.
|
||||
if (IROp->Op == OP_GUESTOPCODE) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Static registers must be consistent at SRA load/store. Evict to ensure.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
@@ -572,23 +534,20 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
Ref Old = Class->RegToSSA[Reg.Reg];
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "RegToSSA invariant");
|
||||
LOGMAN_THROW_A_FMT(IsOld(Node), "Haven't remapped this instruction");
|
||||
|
||||
if (Old != Node) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
Ref Copy;
|
||||
|
||||
if (Reg.Class == FPRFixedClass) {
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Old);
|
||||
Copy = IREmit->_VMov(Header->Size, Map(Old));
|
||||
Copy = IREmit->_VMov(Header->Size, OrderedNodeWrapper::FromImmediate(Reg.Raw));
|
||||
} else {
|
||||
Copy = IREmit->_Copy(Map(Old));
|
||||
Copy = IREmit->_Copy(OrderedNodeWrapper::FromImmediate(Reg.Raw));
|
||||
}
|
||||
|
||||
Remap(Old, Copy);
|
||||
FreeReg(Reg);
|
||||
AssignReg(IR->GetOp<IROp_Header>(Copy), Copy, IROp);
|
||||
RemapReg(Old, PhysicalRegister(Copy));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -604,14 +563,13 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "before remapping");
|
||||
|
||||
if (!IsInRegisterFile(Old)) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
Ref Fill = InsertFill(Old);
|
||||
|
||||
Remap(Old, Fill);
|
||||
AssignReg(IR->GetOp<IROp_Header>(Fill), Fill, IROp);
|
||||
RemapReg(Old, PhysicalRegister(Fill));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -621,20 +579,23 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Ref Node = IR->GetNode(IROp->Args[s]);
|
||||
auto ID = IR->GetID(Node).Value;
|
||||
auto Reg = SSAToReg[ID];
|
||||
|
||||
SourceIndex--;
|
||||
LOGMAN_THROW_A_FMT(SourceIndex >= 0, "Consistent source count");
|
||||
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
auto Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
if (!Reg.IsInvalid()) {
|
||||
IROp->Args[s].SetImmediate(Reg.Raw);
|
||||
|
||||
if (!Reg.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file");
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Node), "sources in file");
|
||||
FreeReg(Reg);
|
||||
}
|
||||
}
|
||||
|
||||
NextUses[IROp->Args[s].ID().Value] = SourcesNextUses[SourceIndex];
|
||||
NextUses[ID] = SourcesNextUses[SourceIndex];
|
||||
}
|
||||
|
||||
// Assign destinations.
|
||||
@@ -642,38 +603,28 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
AssignReg(IROp, CodeNode, IROp);
|
||||
}
|
||||
|
||||
// Remap sources last, since AssignReg can shuffle.
|
||||
if (!SSAToNewSSA.empty()) {
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
Ref Remapped = SSAToNewSSA[IROp->Args[s].ID().Value];
|
||||
|
||||
if (Remapped != nullptr) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, s, Remapped);
|
||||
}
|
||||
}
|
||||
if (IsTrivial(CodeNode, IROp)) {
|
||||
// Delete instructions that only exist for RA
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
} else if (LastNode && TryPostRAMerge(LastNode, CodeNode, IROp)) {
|
||||
// Merge adjacent instructions
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
IREmit->RemovePostRA(LastNode);
|
||||
LastNode = nullptr;
|
||||
} else {
|
||||
LastNode = CodeNode;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(IP >= 1, "IP relative to end of block, iterating forward");
|
||||
--IP;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(SourceIndex == 0, "Consistent source count in block");
|
||||
}
|
||||
|
||||
/* Now that we're done growing things, we can finalize our results.
|
||||
*
|
||||
* TODO: Rework RegisterAllocationData to remove this memcpy, it's pointless.
|
||||
*/
|
||||
AllocData = RegisterAllocationData::Create(SSAToReg.size());
|
||||
AllocData->SpillSlotCount = SpillSlotCount;
|
||||
memcpy(AllocData->Map, SSAToReg.data(), sizeof(PhysicalRegister) * SSAToReg.size());
|
||||
|
||||
PreferredReg.clear();
|
||||
SSAToNewSSA.clear();
|
||||
NewSSAToSSA.clear();
|
||||
SSAToReg.clear();
|
||||
SpillSlots.clear();
|
||||
NextUses.clear();
|
||||
|
||||
IR->GetHeader()->PostRA = true;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<IR::RegisterAllocationPass> CreateRegisterAllocationPass() {
|
||||
|
||||
@@ -12,8 +12,6 @@ $end_info$
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
struct RegisterAllocationDataDeleter;
|
||||
struct RegisterClassType;
|
||||
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
@@ -22,16 +20,6 @@ public:
|
||||
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs;
|
||||
|
||||
/**
|
||||
* @brief Returns the register and class map array
|
||||
*/
|
||||
virtual RegisterAllocationData* GetAllocationData() = 0;
|
||||
|
||||
/**
|
||||
* @brief Returns and transfers ownership of the register and class map array
|
||||
*/
|
||||
virtual std::unique_ptr<RegisterAllocationData, RegisterAllocationDataDeleter> PullAllocationData() = 0;
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -24,6 +24,12 @@ union PhysicalRegister {
|
||||
: Reg(Reg)
|
||||
, Class(Class.Val) {}
|
||||
|
||||
PhysicalRegister(OrderedNodeWrapper Arg)
|
||||
: Raw(Arg.GetImmediate()) {}
|
||||
|
||||
PhysicalRegister(Ref Node)
|
||||
: Raw(Node->Reg) {}
|
||||
|
||||
static const PhysicalRegister Invalid() {
|
||||
return PhysicalRegister(InvalidClass, InvalidReg);
|
||||
}
|
||||
@@ -35,61 +41,4 @@ union PhysicalRegister {
|
||||
|
||||
static_assert(sizeof(PhysicalRegister) == 1);
|
||||
|
||||
struct RegisterAllocationDataDeleter;
|
||||
|
||||
// This class is serialized, can't have any holes in the structure
|
||||
// otherwise ASAN complains about reading uninitialized memory
|
||||
class FEX_PACKED RegisterAllocationData {
|
||||
public:
|
||||
uint32_t SpillSlotCount {};
|
||||
uint32_t MapCount {};
|
||||
PhysicalRegister Map[0];
|
||||
|
||||
PhysicalRegister GetNodeRegister(NodeID Node) const {
|
||||
return Map[Node.Value];
|
||||
}
|
||||
uint32_t SpillSlots() const {
|
||||
return SpillSlotCount;
|
||||
}
|
||||
|
||||
static size_t Size(uint32_t NodeCount) {
|
||||
return sizeof(RegisterAllocationData) + NodeCount * sizeof(Map[0]);
|
||||
}
|
||||
|
||||
using UniquePtr = std::unique_ptr<FEXCore::IR::RegisterAllocationData, RegisterAllocationDataDeleter>;
|
||||
|
||||
static UniquePtr Create(uint32_t NodeCount);
|
||||
|
||||
UniquePtr CreateCopy() const;
|
||||
|
||||
void Serialize(FEXCore::Context::AOTIRWriter& stream) const {
|
||||
stream.Write((const char*)&SpillSlotCount, sizeof(SpillSlotCount));
|
||||
stream.Write((const char*)&MapCount, sizeof(MapCount));
|
||||
// RAData (inline)
|
||||
stream.Write((const char*)&Map[0], sizeof(Map[0]) * MapCount);
|
||||
}
|
||||
};
|
||||
|
||||
struct RegisterAllocationDataDeleter {
|
||||
void operator()(RegisterAllocationData* r) const {
|
||||
FEXCore::Allocator::free(r);
|
||||
}
|
||||
};
|
||||
|
||||
inline auto RegisterAllocationData::Create(uint32_t NodeCount) -> UniquePtr {
|
||||
auto Ret = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(NodeCount));
|
||||
memset(&Ret->Map[0], PhysicalRegister::Invalid().Raw, NodeCount);
|
||||
Ret->SpillSlotCount = 0;
|
||||
Ret->MapCount = NodeCount;
|
||||
return UniquePtr {Ret};
|
||||
}
|
||||
|
||||
inline auto RegisterAllocationData::CreateCopy() const -> UniquePtr {
|
||||
auto copy = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(MapCount));
|
||||
memcpy((void*)©->Map[0], (void*)&Map[0], MapCount * sizeof(Map[0]));
|
||||
copy->SpillSlotCount = SpillSlotCount;
|
||||
copy->MapCount = MapCount;
|
||||
return UniquePtr {copy};
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -265,34 +265,29 @@ public:
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_fundamental_v<TT>)
|
||||
T operator()() const {
|
||||
T operator()() const requires (std::is_fundamental_v<T>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
const T& operator()() const {
|
||||
const fextl::string& operator()() const requires (std::is_same_v<T, fextl::string>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value<T>(T Value) {
|
||||
Value(T Value) requires (!std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
{
|
||||
ValueData = std::move(Value);
|
||||
}
|
||||
|
||||
// Array value types.
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) {
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
{
|
||||
GetListIfExists(Option, &ValueData);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
DefaultValues::Type::StringArrayType& All() {
|
||||
DefaultValues::Type::StringArrayType& All() requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
|
||||
@@ -171,7 +171,7 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
@@ -181,7 +181,7 @@ public:
|
||||
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
|
||||
/**
|
||||
* @brief Checks if a PC is inside of a thread's JIT code buffer.
|
||||
* @brief Checks if a PC is inside any code buffer used by the thread's JIT.
|
||||
*
|
||||
* @param Thread Which thread's code buffers to check inside of.
|
||||
* @param Address The PC to check against.
|
||||
|
||||
@@ -17,7 +17,6 @@ namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
enum class SyscallFlags : uint8_t {
|
||||
DEFAULT = 0,
|
||||
|
||||
@@ -53,9 +53,9 @@ namespace Throw {
|
||||
MFmt(fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt(pred, __VA_ARGS__); \
|
||||
#define LOGMAN_THROW_A_FMT(pred, format, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt((pred), "{}:{}, {}: " format, __FILE_NAME__, __LINE__, __FUNCTION__ __VA_OPT__(, ) __VA_ARGS__); \
|
||||
} while (0)
|
||||
#else
|
||||
static inline void AFmt(bool, const char*, ...) {}
|
||||
|
||||
@@ -44,6 +44,13 @@ public:
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
|
||||
// Asserts that the mutex isn't exclusively owned by the calling thread.
|
||||
void check_lock_owned_by_self() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == EDEADLK, "User of unique lock must have already locked mutex as write!");
|
||||
}
|
||||
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
@@ -85,6 +92,13 @@ public:
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
// Asserts that the rwlock isn't exclusively owned by the calling thread.
|
||||
void check_lock_owned_by_self_as_write() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == EDEADLK, "User of rwlock must have already locked mutex as write!");
|
||||
}
|
||||
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
|
||||
@@ -24,7 +24,7 @@ namespace FEXCore::Utils {
|
||||
* - This is relatively cheap.
|
||||
* - `Unclaim` when the buffer won't be used again for an extended period.
|
||||
* - This is expensive and requires a mutex shared between threads
|
||||
* - `FixedSizePooledAllocation` helper class provided to help with this.
|
||||
* - `PoolBufferWithTimedRetirement` helper class provided to help with this.
|
||||
*
|
||||
* Once the client has disowned a buffer then the allocator is free to reclaim the buffer when another thread is trying to `Claim` a new buffer.
|
||||
* The buffer getting claimed from a disowned client must have had its last use greater than the defined `DURATION` before it has a chance to get
|
||||
@@ -124,7 +124,7 @@ public:
|
||||
*
|
||||
* @param Buffer - The iterator that was previously given with ClaimBuffer
|
||||
*/
|
||||
void UnclaimBuffer(ContainerType::iterator Buffer, BufferOwnedFlag* ClientFlag) {
|
||||
void UnclaimBuffer(const ContainerType::iterator& Buffer, BufferOwnedFlag* ClientFlag) {
|
||||
// Transition the buffer to free, unclaiming if it wasn't free prior.
|
||||
if (ClientFlag->exchange(ClientFlags::FLAG_FREE) != ClientFlags::FLAG_FREE) {
|
||||
std::unique_lock lk {AllocationMutex};
|
||||
@@ -149,24 +149,45 @@ public:
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Try to reown a buffer that we have previous disowned, failing that, claim a new buffer
|
||||
* @brief Try to reown a buffer that was previously disowned
|
||||
*
|
||||
* @param Buffer - The buffer we previously disowned
|
||||
* @param Size - The size of the buffer
|
||||
* @param CurrentClientFlag - The client tracked flag
|
||||
*
|
||||
* Once a DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always Reown a buffer after disowning it before use!
|
||||
* Once DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always reown a buffer before use!
|
||||
*
|
||||
* @return Either the original buffer passed in if we managed to reclaim, or a new buffer if we couldn't
|
||||
* @return The original buffer passed in on successful reown, otherwise std::nullopt
|
||||
*/
|
||||
ContainerType::iterator ReownOrClaimBuffer(ContainerType::iterator Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
std::optional<ContainerType::iterator> TryToReownBuffer(const ContainerType::iterator& Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
ClientFlags Expected = ClientFlags::FLAG_DISOWNED;
|
||||
if (CurrentClientFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_OWNED)) {
|
||||
// If we managed to change the flag from DISOWNED to OWNED then we have successfully reclaimed
|
||||
// Finish setting up state
|
||||
(*Buffer)->LastUsed.store(ClockType::now(), std::memory_order_relaxed);
|
||||
return Buffer;
|
||||
if (!CurrentClientFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_OWNED)) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
// If we managed to change the flag from DISOWNED to OWNED then we have successfully reclaimed
|
||||
// Finish setting up state
|
||||
(*Buffer)->LastUsed.store(ClockType::now(), std::memory_order_relaxed);
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Try to reown a buffer that was previously disowned, failing that, claim a new buffer
|
||||
*
|
||||
* @param Buffer - The buffer we previously disowned
|
||||
* @param Size - The size of the buffer
|
||||
* @param CurrentClientFlag - The client tracked flag
|
||||
*
|
||||
* Once DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always reown a buffer before use!
|
||||
*
|
||||
* @return The original buffer passed in on successful reown, otherwise a new buffer
|
||||
*/
|
||||
ContainerType::iterator ReownOrClaimBuffer(const ContainerType::iterator& Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
auto Reowned = TryToReownBuffer(Buffer, Size, CurrentClientFlag);
|
||||
if (Reowned) {
|
||||
return Reowned.value();
|
||||
}
|
||||
|
||||
// Couldn't reclaim, just get a new buffer
|
||||
@@ -410,7 +431,7 @@ private:
|
||||
* - Frees stale buffers opportunistically
|
||||
*/
|
||||
template<typename Type, size_t PeriodMS, size_t PeriodFrequency>
|
||||
class FixedSizePooledAllocation final {
|
||||
class PoolBufferWithTimedRetirement final {
|
||||
// If the delayed object reclaimer is more than the thread pool allocator's duration then the pool allocator would always need to reclaim
|
||||
// the buffer rather than giving it back.
|
||||
static_assert(std::chrono::duration(std::chrono::milliseconds(PeriodMS)) <= IntrusivePooledAllocator::DURATION, "DeplayedObjectReclaimer "
|
||||
@@ -420,24 +441,39 @@ class FixedSizePooledAllocation final {
|
||||
"duration");
|
||||
|
||||
public:
|
||||
FixedSizePooledAllocation(IntrusivePooledAllocator& Allocator, size_t Size)
|
||||
PoolBufferWithTimedRetirement(IntrusivePooledAllocator& Allocator, size_t Size)
|
||||
: ThreadAllocator {Allocator}
|
||||
, Size {Size} {}
|
||||
|
||||
/**
|
||||
* @brief Return the owned buffer or allocate another one from the `Allocator`
|
||||
*
|
||||
* The buffer returned isn't guaranteed to be the exact size of `Size` but it will be at least `Size`.
|
||||
* The contents of the memory returned isn't guaranteed to be zero initialized or not.
|
||||
* Not even guaranteed to contain the previous data from the previous reowning if the pointer is the same.
|
||||
* The buffer is guaranteed to have at least `Size` bytes of data.
|
||||
* The initial data in the buffer is undefined, even when the buffer is just reowned.
|
||||
*
|
||||
* @return object of type `Type` allocated with at least the size of `Size` from the constructor
|
||||
* @param NewSize Optional new size for managed data
|
||||
*
|
||||
* @return object of type `Type` allocated within the selected buffer
|
||||
*/
|
||||
Type ReownOrClaimBuffer() {
|
||||
if (!FEXCore::Utils::IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag)) {
|
||||
Info = ThreadAllocator.ReownOrClaimBuffer(Info, Size, &ClientOwnedFlag);
|
||||
Type ReownOrClaimBuffer(std::optional<size_t> NewSize = std::nullopt) {
|
||||
// Check if we can cheaply re-own a previous buffer
|
||||
std::optional Buffer =
|
||||
IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag) ? Info : ThreadAllocator.TryToReownBuffer(Info, Size, &ClientOwnedFlag);
|
||||
|
||||
// Ensure the now owned buffer has enough space. If not, unclaim it and proceed to claim a new one
|
||||
if (NewSize && Buffer && (**Buffer)->Size < NewSize.value()) {
|
||||
UnclaimBuffer();
|
||||
Buffer.reset();
|
||||
}
|
||||
|
||||
// Claim a new buffer if needed
|
||||
Size = NewSize.value_or(Size);
|
||||
if (!Buffer) {
|
||||
Buffer = ThreadAllocator.ClaimBuffer(Size, &ClientOwnedFlag);
|
||||
}
|
||||
|
||||
Info = *Buffer;
|
||||
|
||||
// Putting a memset here is very handy for using thread sanitizer to find buffer usage races
|
||||
// Leaving this here for future excavation that will definitely occur here
|
||||
// memset((*Info)->Ptr, 0, Size);
|
||||
|
||||
@@ -25,6 +25,9 @@ struct default_delete : public std::default_delete<T> {
|
||||
template<class T, class Deleter = fextl::default_delete<T>>
|
||||
using unique_ptr = std::unique_ptr<T, Deleter>;
|
||||
|
||||
template<class T>
|
||||
using shared_ptr = std::shared_ptr<T>;
|
||||
|
||||
template<class T, class... Args>
|
||||
requires (!std::is_array_v<T>)
|
||||
fextl::unique_ptr<T> make_unique(Args&&... args) {
|
||||
@@ -32,4 +35,10 @@ fextl::unique_ptr<T> make_unique(Args&&... args) {
|
||||
auto Result = ::new (ptr) T(std::forward<Args>(args)...);
|
||||
return fextl::unique_ptr<T>(Result);
|
||||
}
|
||||
|
||||
template<class T, class... Args>
|
||||
requires (!std::is_array_v<T>)
|
||||
fextl::shared_ptr<T> make_shared(Args&&... args) {
|
||||
return std::allocate_shared<T>(fextl::FEXAlloc<T> {}, std::forward<Args>(args)...);
|
||||
}
|
||||
} // namespace fextl
|
||||
@@ -84,7 +84,8 @@ def IsSupportedDistro():
|
||||
# We only support what is available in ppa:fex-emu/fex
|
||||
return Distro[1] == "22.04" or \
|
||||
Distro[1] == "24.04" or \
|
||||
Distro[1] == "24.10"
|
||||
Distro[1] == "24.10" or \
|
||||
Distro[1] == "25.04"
|
||||
|
||||
return False
|
||||
|
||||
|
||||
+30
-9
@@ -54,6 +54,7 @@ private:
|
||||
std::vector<pollfd> PollFDs;
|
||||
std::optional<int> CurrentFD; // FD that is currently being processed
|
||||
|
||||
bool is_stopped = false;
|
||||
int AsyncStopRequest[2] = {-1, -1};
|
||||
|
||||
// Maps FD to callback
|
||||
@@ -91,12 +92,22 @@ public:
|
||||
::close(AsyncStopRequest[1]);
|
||||
}
|
||||
|
||||
error run(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
void cleanup() {
|
||||
callbacks.clear();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool stopped() const {
|
||||
return is_stopped;
|
||||
}
|
||||
|
||||
error run_one(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
// Process events queued before entering wait loop
|
||||
update_fd_list();
|
||||
|
||||
timespec ts = to_timespec(Timeout.value_or(std::chrono::nanoseconds {0}));
|
||||
|
||||
// ppoll may return EINTR/EAGAIN, so a loop is used here. Normally, we return in the first iteration.
|
||||
while (true) {
|
||||
int Result = ::ppoll(PollFDs.data(), PollFDs.size(), Timeout ? &ts : nullptr, nullptr);
|
||||
|
||||
@@ -104,14 +115,10 @@ public:
|
||||
if (errno == EINTR || errno == EAGAIN) {
|
||||
continue;
|
||||
}
|
||||
callbacks.clear();
|
||||
return error::generic_errno;
|
||||
} else if (Result == 0) {
|
||||
callbacks.clear();
|
||||
return error::timeout;
|
||||
} else {
|
||||
bool exit_requested = false;
|
||||
|
||||
// Walk the FDs and see if we got any results
|
||||
for (auto& ActiveFD : PollFDs) {
|
||||
if (ActiveFD.revents == 0) {
|
||||
@@ -139,14 +146,14 @@ public:
|
||||
ActiveFD.revents = 0;
|
||||
}
|
||||
} else if (Ret == post_callback::stop_reactor) {
|
||||
exit_requested = true;
|
||||
is_stopped = true;
|
||||
}
|
||||
CurrentFD.reset();
|
||||
}
|
||||
if (ActiveFD.revents & (POLLHUP | POLLERR | POLLNVAL | POLLRDHUP)) {
|
||||
auto Callback = std::move(callbacks[ActiveFD.fd]);
|
||||
if (Callback) {
|
||||
exit_requested |= (Callback(error::eof) == post_callback::stop_reactor);
|
||||
is_stopped |= (Callback(error::eof) == post_callback::stop_reactor);
|
||||
}
|
||||
// Error or hangup, erase the socket from our list
|
||||
QueuedEvents.push_back(Event {.FD = {.fd = ActiveFD.fd}, .Erase = true});
|
||||
@@ -155,12 +162,23 @@ public:
|
||||
ActiveFD.revents = 0;
|
||||
}
|
||||
|
||||
if (exit_requested) {
|
||||
callbacks.clear();
|
||||
if (is_stopped) {
|
||||
cleanup();
|
||||
return error::success;
|
||||
}
|
||||
|
||||
update_fd_list();
|
||||
return error::success;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
error run(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
while (true) {
|
||||
auto Result = run_one(Timeout);
|
||||
if (Result != error::success || is_stopped) {
|
||||
cleanup();
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -404,6 +422,9 @@ struct posix_descriptor {
|
||||
, FD(std::exchange(Other.FD, -1)) {}
|
||||
|
||||
posix_descriptor& operator=(posix_descriptor&& Other) {
|
||||
if (&Other == this) {
|
||||
return *this;
|
||||
}
|
||||
posix_descriptor::~posix_descriptor();
|
||||
Reactor = Other.Reactor;
|
||||
FD = std::exchange(Other.FD, -1);
|
||||
|
||||
@@ -152,6 +152,10 @@ public:
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char* ConfigName, const char* ConfigString);
|
||||
void SetCurrentConfigFile(const fextl::string& Filename) {
|
||||
CurrentConfigFile = Filename;
|
||||
}
|
||||
fextl::string CurrentConfigFile;
|
||||
};
|
||||
|
||||
class MainLoader final : public OptionMapper {
|
||||
@@ -199,6 +203,7 @@ void OptionMapper::MapNameToOption(const char* ConfigName, const char* ConfigStr
|
||||
}
|
||||
|
||||
if (!KeyOptionValue.has_value()) {
|
||||
LogMan::Msg::IFmt("Unknown configuration option '{}' in JSON config file '{}'", ConfigName, CurrentConfigFile);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -223,6 +228,7 @@ MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigF
|
||||
, Config {ConfigFile} {}
|
||||
|
||||
void MainLoader::Load() {
|
||||
SetCurrentConfigFile(Config);
|
||||
JSON::LoadJSonConfig(Config, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
}
|
||||
|
||||
@@ -236,6 +242,7 @@ AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType T
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
SetCurrentConfigFile(Config);
|
||||
JSON::LoadJSonConfig(Config, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <cstdlib>
|
||||
#include <fcntl.h>
|
||||
#include <linux/limits.h>
|
||||
#include <unistd.h>
|
||||
@@ -23,6 +24,7 @@
|
||||
#include <sys/un.h>
|
||||
#include <sys/uio.h>
|
||||
#include <thread>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXServerClient {
|
||||
int RequestPIDFDPacket(int ServerSocket, PacketType Type) {
|
||||
@@ -222,7 +224,14 @@ int ConnectToAndStartServer(std::string_view InterpreterPath) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
fextl::string FEXServerPath = fextl::fmt::format("{}/FEXServer", InterpreterPath);
|
||||
// Extract directory from InterpreterPath
|
||||
fextl::string InterpreterDir {InterpreterPath};
|
||||
size_t LastSlash = InterpreterDir.rfind('/');
|
||||
if (LastSlash != fextl::string::npos) {
|
||||
InterpreterDir = InterpreterDir.substr(0, LastSlash);
|
||||
}
|
||||
|
||||
fextl::string FEXServerPath = fextl::fmt::format("{}/FEXServer", InterpreterDir);
|
||||
// Check if a local FEXServer next to FEXInterpreter exists
|
||||
// If it does then it takes priority over the installed one
|
||||
if (!FHU::Filesystem::Exists(FEXServerPath)) {
|
||||
|
||||
@@ -434,6 +434,35 @@ public:
|
||||
private:
|
||||
fextl::vector<std::pair<std::string_view, std::string_view>> Env;
|
||||
};
|
||||
|
||||
class SimpleSyscallHandler : public FEXCore::HLE::SyscallHandler, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
SimpleSyscallHandler() {
|
||||
// Just claim to be linux 64-bit for simplicity.
|
||||
OSABI = FEXCore::HLE::SyscallOSABI::OS_LINUX64;
|
||||
}
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) override {
|
||||
// Don't do anything
|
||||
return 0;
|
||||
}
|
||||
|
||||
FEXCore::HLE::SyscallABI GetSyscallABI(uint64_t Syscall) override {
|
||||
if (Syscall == 0) {
|
||||
// Claim syscall 0 is simple for instcountci inline tests.
|
||||
return FEXCore::HLE::SyscallABI {
|
||||
.NumArgs = 0,
|
||||
.HasReturn = true,
|
||||
.HostSyscallNumber = 0, // Just map to host syscall zero, it isn't going to get called.
|
||||
};
|
||||
}
|
||||
return {0, false, -1};
|
||||
}
|
||||
|
||||
// These are no-ops implementations of the SyscallHandler API
|
||||
FEXCore::HLE::AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) override {
|
||||
return {0, 0};
|
||||
}
|
||||
};
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv, char** const envp) {
|
||||
@@ -620,7 +649,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
auto SignalDelegation = FEX::DummyHandlers::CreateSignalDelegator();
|
||||
auto SyscallHandler = FEX::DummyHandlers::CreateSyscallHandler();
|
||||
auto SyscallHandler = fextl::make_unique<SimpleSyscallHandler>();
|
||||
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
|
||||
@@ -406,13 +406,8 @@ public:
|
||||
}
|
||||
|
||||
uint64_t GetStackPointer() override {
|
||||
if (Config.Is64BitMode()) {
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(StackSize())) + StackSize();
|
||||
} else {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), StackSize()));
|
||||
LOGMAN_THROW_A_FMT(Result != ~0ULL, "Stack Pointer mmap failed");
|
||||
return Result + StackSize();
|
||||
}
|
||||
LOGMAN_MSG_A_FMT("This should be unused.");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
uint64_t DefaultRIP() const override {
|
||||
@@ -460,6 +455,11 @@ public:
|
||||
DoMMap(region, size);
|
||||
}
|
||||
|
||||
if (!Config.Is64BitMode()) {
|
||||
// 32-bit gets a fixed page allocated for stack.
|
||||
DoMMap(STACK_OFFSET, StackSize());
|
||||
}
|
||||
|
||||
LoadMemory();
|
||||
|
||||
return true;
|
||||
|
||||
@@ -10,8 +10,6 @@ target_link_libraries(FEXBash
|
||||
FEXCore
|
||||
Common
|
||||
JemallocLibs
|
||||
LinuxEmulation
|
||||
${PTHREAD_LIB}
|
||||
)
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
|
||||
@@ -8,44 +8,22 @@ $end_info$
|
||||
|
||||
#include "ConfigDefines.h"
|
||||
#include "Common/ArgumentLoader.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FEXServerClient.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <filesystem>
|
||||
#include <iterator>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <string>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
|
||||
int main(int argc, char** argv, char** const envp) {
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateMainLayer());
|
||||
FEX::ArgLoader::ArgLoader ArgsLoader(FEX::ArgLoader::ArgLoader::LoadType::WITHOUT_FEXLOADER_PARSER, argc, argv);
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
// Reload the meta layer
|
||||
FEXCore::Config::ReloadMetaLayer();
|
||||
|
||||
auto Args = ArgsLoader.Get();
|
||||
|
||||
// Ensure FEXServer is setup before config options try to pull CONFIG_ROOTFS
|
||||
if (!FEXServerClient::SetupClient(argv[0])) {
|
||||
LogMan::Msg::EFmt("FEXServerClient: Failure to setup client");
|
||||
return -1;
|
||||
}
|
||||
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
std::vector<const char*> Argv;
|
||||
fextl::string BinShPath = RootFSPath() + "/bin/sh";
|
||||
fextl::string BinBashPath = RootFSPath() + "/bin/bash";
|
||||
// FEXInterpreter will handle finding bash in the rootfs
|
||||
// Use /bin/sh for -c commands and /bin/bash for interactive mode
|
||||
const char* BashPath = Args.empty() ? "/bin/bash" : "/bin/sh";
|
||||
|
||||
std::string FEXInterpreterPath = std::filesystem::path(argv[0]).parent_path().string() + "FEXInterpreter";
|
||||
// Check if a local FEXInterpreter to FEXBash exists
|
||||
@@ -55,7 +33,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
const char* FEXArgs[] = {
|
||||
FEXInterpreterPath.c_str(),
|
||||
Args.empty() ? BinBashPath.c_str() : BinShPath.c_str(),
|
||||
BashPath,
|
||||
"-c",
|
||||
};
|
||||
|
||||
|
||||
@@ -74,7 +74,7 @@ class ELFCodeLoader final : public FEX::CodeLoader {
|
||||
|
||||
void* rv = Handler->GuestMmap(nullptr, (void*)addr, size, prot, flags, file.fd, off);
|
||||
|
||||
if (rv == MAP_FAILED) {
|
||||
if (FEX::HLE::HasSyscallError(rv)) {
|
||||
// uhoh, something went wrong
|
||||
LogMan::Msg::EFmt("MapFile: Some elf mapping failed, {}, fd: {}\n", errno, file.fd);
|
||||
return false;
|
||||
@@ -120,7 +120,7 @@ class ELFCodeLoader final : public FEX::CodeLoader {
|
||||
auto TotalSize = CalculateTotalElfSize(Elf.phdrs) + (BrkBase ? BRK_SIZE : 0);
|
||||
LoadBase =
|
||||
(uintptr_t)Handler->GuestMmap(nullptr, reinterpret_cast<void*>(LoadHint), TotalSize, PROT_NONE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
if ((void*)LoadBase == MAP_FAILED) {
|
||||
if (FEX::HLE::HasSyscallError(LoadBase)) {
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -155,7 +155,7 @@ class ELFCodeLoader final : public FEX::CodeLoader {
|
||||
|
||||
if (BSSPageStart != BSSPageEnd) {
|
||||
auto bss = Handler->GuestMmap(nullptr, (void*)BSSPageStart, BSSPageEnd - BSSPageStart, MapProt, MapType | MAP_ANONYMOUS, -1, 0);
|
||||
if ((void*)bss == MAP_FAILED) {
|
||||
if (FEX::HLE::HasSyscallError(bss)) {
|
||||
LogMan::Msg::EFmt("Failed to allocate BSS @ {}, {}\n", fmt::ptr(bss), errno);
|
||||
return {};
|
||||
}
|
||||
@@ -409,7 +409,8 @@ public:
|
||||
StackPointerBase = Handler->GuestMmap(nullptr, reinterpret_cast<void*>(StackHint), FULL_STACK_SIZE, PROT_NONE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN | MAP_NORESERVE, -1, 0);
|
||||
|
||||
if (StackPointerBase == reinterpret_cast<void*>(~0ULL)) {
|
||||
|
||||
if (FEX::HLE::HasSyscallError(StackPointerBase)) {
|
||||
LogMan::Msg::EFmt("Allocating stack failed");
|
||||
return false;
|
||||
}
|
||||
@@ -419,7 +420,7 @@ public:
|
||||
Handler->GuestMmap(nullptr, reinterpret_cast<void*>(reinterpret_cast<uint64_t>(StackPointerBase) + FULL_STACK_SIZE - StackSize()),
|
||||
StackSize(), PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
|
||||
|
||||
if (StackPointer == ~0ULL) {
|
||||
if (FEX::HLE::HasSyscallError(StackPointer)) {
|
||||
LogMan::Msg::EFmt("Allocating stack failed");
|
||||
return false;
|
||||
}
|
||||
@@ -542,7 +543,7 @@ public:
|
||||
BrkStart =
|
||||
(uint64_t)Handler->GuestMmap(nullptr, (void*)BrkBase, BRK_SIZE, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
if ((void*)BrkStart == MAP_FAILED) {
|
||||
if (FEX::HLE::HasSyscallError(BrkStart)) {
|
||||
LogMan::Msg::EFmt("Failed to allocate BRK @ {:x}, {}\n", BrkBase, errno);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -24,6 +24,7 @@ constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
int ServerLockFD {-1};
|
||||
std::optional<fasio::tcp_acceptor> ServerAcceptor;
|
||||
std::optional<fasio::tcp_acceptor> ServerFSAcceptor;
|
||||
int NumClients = 0;
|
||||
time_t RequestTimeout {10};
|
||||
bool Foreground {false};
|
||||
std::vector<struct pollfd> PollFDs {};
|
||||
@@ -210,6 +211,7 @@ bool InitializeServerSocket(bool abstract) {
|
||||
}
|
||||
|
||||
int FD = Socket->FD;
|
||||
++NumClients;
|
||||
Reactor.bind_handler(
|
||||
pollfd {
|
||||
.fd = FD,
|
||||
@@ -219,6 +221,7 @@ bool InitializeServerSocket(bool abstract) {
|
||||
[Socket = std::move(Socket).value()](fasio::error ec) mutable {
|
||||
if (ec != fasio::error::success) {
|
||||
close(Socket.FD);
|
||||
--NumClients;
|
||||
return fasio::post_callback::drop;
|
||||
}
|
||||
HandleSocketData(Socket);
|
||||
@@ -389,7 +392,18 @@ void CloseConnections() {
|
||||
|
||||
void WaitForRequests() {
|
||||
Reactor.enable_async_stop();
|
||||
Reactor.run(Foreground ? std::nullopt : std::optional {std::chrono::seconds {RequestTimeout}});
|
||||
|
||||
while (true) {
|
||||
std::optional Timeout = std::chrono::seconds {RequestTimeout};
|
||||
if (Foreground || NumClients > 0) {
|
||||
Timeout.reset();
|
||||
}
|
||||
auto Result = Reactor.run_one(Timeout);
|
||||
if (Result != fasio::error::success || Reactor.stopped()) {
|
||||
Reactor.cleanup();
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("[FEXServer] Shutting Down");
|
||||
|
||||
|
||||
@@ -736,8 +736,7 @@ uint64_t SyscallHandler::HandleBRK(FEXCore::Core::CpuStateFrame* Frame, void* Ad
|
||||
NewBRK = (uint64_t)GuestMmap(Frame->Thread, (void*)(DataSpace + DataSpaceMaxSize), AllocateNewSize, PROT_READ | PROT_WRITE,
|
||||
MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
|
||||
if (NewBRK != ~0ULL && NewBRK != (DataSpace + DataSpaceMaxSize)) {
|
||||
if (!FEX::HLE::HasSyscallError(NewBRK) && NewBRK != (DataSpace + DataSpaceMaxSize)) {
|
||||
// Couldn't allocate that the region we wanted
|
||||
// Can happen if MAP_FIXED_NOREPLACE isn't understood by the kernel
|
||||
[[maybe_unused]] int ok = GuestMunmap(Frame->Thread, reinterpret_cast<void*>(NewBRK), AllocateNewSize);
|
||||
@@ -745,7 +744,7 @@ uint64_t SyscallHandler::HandleBRK(FEXCore::Core::CpuStateFrame* Frame, void* Ad
|
||||
NewBRK = ~0ULL;
|
||||
}
|
||||
|
||||
if (NewBRK == ~0ULL) {
|
||||
if (FEX::HLE::HasSyscallError(NewBRK)) {
|
||||
// If we couldn't allocate a new region then out of memory
|
||||
return DataSpace + DataSpaceSize;
|
||||
} else {
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "LinuxSyscalls/LinuxAllocator.h"
|
||||
#include "LinuxSyscalls/ThreadManager.h"
|
||||
#include "LinuxSyscalls/Seccomp/SeccompEmulator.h"
|
||||
#include "LinuxSyscalls/SyscallsVMATracking.h"
|
||||
#include "ArchHelpers/MContext.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -22,6 +23,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -233,25 +235,57 @@ public:
|
||||
return Version & 0xFFFF;
|
||||
}
|
||||
|
||||
FEX::HLE::MemAllocator* Get32BitAllocator() {
|
||||
virtual FEX::HLE::MemAllocator* Get32BitAllocator() {
|
||||
return Alloc32Handler.get();
|
||||
}
|
||||
|
||||
// does a mmap as if done via a guest syscall
|
||||
virtual void* GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) = 0;
|
||||
void* GuestMmap(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
|
||||
// does a guest munmap as if done via a guest syscall
|
||||
virtual int GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) = 0;
|
||||
virtual uint64_t GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) = 0;
|
||||
uint64_t GuestMunmap(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length);
|
||||
|
||||
uint64_t GuestMremap(bool Is64Bit, FEXCore::Core::InternalThreadState*, void* old_address, size_t old_size, size_t new_size, int flags,
|
||||
void* new_address);
|
||||
uint64_t GuestMprotect(FEXCore::Core::InternalThreadState*, void* addr, size_t len, int prot);
|
||||
uint64_t GuestShmat(bool Is64Bit, FEXCore::Core::InternalThreadState*, int shmid, const void* shmaddr, int shmflg);
|
||||
uint64_t GuestShmdt(bool Is64Bit, FEXCore::Core::InternalThreadState*, const void* shmaddr);
|
||||
|
||||
///// Memory Manager tracking /////
|
||||
void TrackMmap(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int Prot, int Flags, int fd, off_t Offset);
|
||||
void TrackMunmap(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size);
|
||||
void TrackMprotect(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int Prot);
|
||||
void TrackMremap(FEXCore::Core::InternalThreadState* Thread, uintptr_t OldAddress, size_t OldSize, size_t NewSize, int flags, uintptr_t NewAddress);
|
||||
void TrackShmat(FEXCore::Core::InternalThreadState* Thread, int shmid, uintptr_t Base, int shmflg);
|
||||
void TrackShmdt(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base);
|
||||
void TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
void TrackMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length);
|
||||
void TrackMremap(FEXCore::Core::InternalThreadState* Thread, uint64_t OldAddress, size_t OldSize, size_t NewSize, int flags, uint64_t NewAddress);
|
||||
void TrackShmat(FEXCore::Core::InternalThreadState* Thread, int shmid, uint64_t shmaddr, int shmflg, uint64_t Length);
|
||||
uint64_t TrackShmdt(FEXCore::Core::InternalThreadState* Thread, uint64_t shmaddr);
|
||||
void TrackMprotect(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t len, int prot);
|
||||
void TrackMadvise(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int advice);
|
||||
|
||||
void InvalidateCodeRangeIfNecessary(FEXCore::Core::InternalThreadState* Thread, uint64_t Base, uint64_t Length) {
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
TM.InvalidateGuestCodeRange(Thread, Base, Length);
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidateCodeRangeIfNecessaryOnRemap(FEXCore::Core::InternalThreadState* Thread, uint64_t OldAddress, uint64_t NewAddress,
|
||||
size_t OldSize, size_t NewSize) {
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
if (OldAddress != NewAddress) {
|
||||
if (OldSize != 0) {
|
||||
// This also handles the MREMAP_DONTUNMAP case
|
||||
TM.InvalidateGuestCodeRange(Thread, OldAddress, OldSize);
|
||||
}
|
||||
} else {
|
||||
// If mapping shrunk, flush the unmapped region
|
||||
if (OldSize > NewSize) {
|
||||
TM.InvalidateGuestCodeRange(Thread, OldAddress + NewSize, OldSize - NewSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
///// VMA (Virtual Memory Area) tracking /////
|
||||
static bool HandleSegfault(FEXCore::Core::InternalThreadState* Thread, int Signal, void* info, void* ucontext);
|
||||
void MarkGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
@@ -282,6 +316,8 @@ public:
|
||||
|
||||
constexpr static uint64_t TASK_MAX_64BIT = (1ULL << 48);
|
||||
|
||||
VMATracking::VMATracking VMATracking;
|
||||
|
||||
protected:
|
||||
SyscallHandler(FEXCore::Context::Context* _CTX, FEX::HLE::SignalDelegator* _SignalDelegation, FEX::HLE::ThunkHandler* ThunkHandler);
|
||||
|
||||
@@ -303,7 +339,6 @@ protected:
|
||||
uint32_t GuestKernelVersion {};
|
||||
|
||||
FEXCore::Context::Context* CTX;
|
||||
|
||||
private:
|
||||
|
||||
FEX::HLE::SignalDelegator* SignalDelegation;
|
||||
@@ -323,108 +358,7 @@ private:
|
||||
fextl::unique_ptr<FEXCore::HLE::SourcecodeMap>
|
||||
GenerateMap(const std::string_view& GuestBinaryFile, const std::string_view& GuestBinaryFileId) override;
|
||||
|
||||
///// VMA (Virtual Memory Area) tracking /////
|
||||
|
||||
|
||||
struct SpecialDev {
|
||||
static constexpr uint64_t Anon = 0x1'0000'0000; // Anonymous shared mapping, id is incrementing allocation number
|
||||
static constexpr uint64_t SHM = 0x2'0000'0000; // sys-v shm, id is shmid
|
||||
};
|
||||
|
||||
// Memory Resource ID
|
||||
// An id that can be used to identify when shared mappings actually have the same backing storage
|
||||
// when dev != SpecialDev::Anon, this is unique system wide
|
||||
struct MRID {
|
||||
uint64_t dev; // kernel dev_t is actually 32-bits, we use the extra bits to track SpecialDevs
|
||||
uint64_t id;
|
||||
|
||||
bool operator<(const MRID& other) const {
|
||||
return std::tie(dev, id) < std::tie(other.dev, other.id);
|
||||
}
|
||||
};
|
||||
|
||||
struct VMAEntry;
|
||||
|
||||
// Used to all MAP_SHARED VMAs of a system resource.
|
||||
struct MappedResource {
|
||||
using ContainerType = fextl::map<MRID, MappedResource>;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry* AOTIRCacheEntry;
|
||||
// Pointer to lowest memory range this file is mapped to
|
||||
VMAEntry* FirstVMA;
|
||||
uint64_t Length; // 0 if not fixed size
|
||||
ContainerType::iterator Iterator;
|
||||
};
|
||||
|
||||
union VMAProt {
|
||||
struct {
|
||||
bool Readable : 1;
|
||||
bool Writable : 1;
|
||||
bool Executable : 1;
|
||||
};
|
||||
uint8_t All : 3;
|
||||
|
||||
|
||||
static VMAProt fromProt(int Prot);
|
||||
static VMAProt fromSHM(int SHMFlg);
|
||||
};
|
||||
|
||||
struct VMAFlags {
|
||||
bool Shared : 1;
|
||||
|
||||
static VMAFlags fromFlags(int Flags);
|
||||
};
|
||||
|
||||
struct VMAEntry {
|
||||
MappedResource* Resource;
|
||||
|
||||
// these are for intrusive linked list tracking, starting from Resource->FirstVMA and ordered by address
|
||||
VMAEntry* ResourcePrevVMA;
|
||||
VMAEntry* ResourceNextVMA;
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Offset;
|
||||
uint64_t Length;
|
||||
|
||||
VMAFlags Flags;
|
||||
VMAProt Prot;
|
||||
};
|
||||
|
||||
struct VMATracking {
|
||||
using VMAEntry = SyscallHandler::VMAEntry;
|
||||
// Held while reading/writing this struct
|
||||
FEXCore::ForkableSharedMutex Mutex;
|
||||
|
||||
// Memory ranges indexed by page aligned starting address
|
||||
fextl::map<uint64_t, VMAEntry> VMAs;
|
||||
|
||||
using VMACIterator = decltype(VMAs)::const_iterator;
|
||||
|
||||
MappedResource::ContainerType MappedResources;
|
||||
|
||||
// Mutex must be at least shared_locked before calling
|
||||
VMACIterator LookupVMAUnsafe(uint64_t GuestAddr) const;
|
||||
|
||||
// Mutex must be unique_locked before calling
|
||||
void SetUnsafe(FEXCore::Context::Context* Ctx, MappedResource* MappedResource, uintptr_t Base, uintptr_t Offset, uintptr_t Length,
|
||||
VMAFlags Flags, VMAProt Prot);
|
||||
|
||||
// Mutex must be unique_locked before calling
|
||||
void ClearUnsafe(FEXCore::Context::Context* Ctx, uintptr_t Base, uintptr_t Length, MappedResource* PreservedMappedResource = nullptr);
|
||||
|
||||
// Mutex must be unique_locked before calling
|
||||
void ChangeUnsafe(uintptr_t Base, uintptr_t Length, VMAProt Prot);
|
||||
|
||||
// Mutex must be unique_locked before calling
|
||||
// Returns the Size fo the Shm or 0 if not found
|
||||
uintptr_t ClearShmUnsafe(FEXCore::Context::Context* Ctx, uintptr_t Base);
|
||||
private:
|
||||
bool ListRemove(VMAEntry* Mapping);
|
||||
void ListReplace(VMAEntry* Mapping, VMAEntry* NewMapping);
|
||||
void ListInsertAfter(VMAEntry* Mapping, VMAEntry* NewMapping);
|
||||
void ListPrepend(MappedResource* Resource, VMAEntry* NewVMA);
|
||||
static void ListCheckVMALinks(VMAEntry* VMA);
|
||||
} VMATracking;
|
||||
std::atomic<uint64_t> AnonSharedId {1};
|
||||
};
|
||||
|
||||
uint64_t HandleSyscall(SyscallHandler* Handler, FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args);
|
||||
|
||||
@@ -22,30 +22,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
|
||||
/// Helpers ///
|
||||
auto SyscallHandler::VMAProt::fromProt(int Prot) -> VMAProt {
|
||||
return VMAProt {
|
||||
.Readable = (Prot & PROT_READ) != 0,
|
||||
.Writable = (Prot & PROT_WRITE) != 0,
|
||||
.Executable = (Prot & PROT_EXEC) != 0,
|
||||
};
|
||||
}
|
||||
|
||||
auto SyscallHandler::VMAProt::fromSHM(int SHMFlg) -> VMAProt {
|
||||
return VMAProt {
|
||||
.Readable = true,
|
||||
.Writable = SHMFlg & SHM_RDONLY ? false : true,
|
||||
.Executable = SHMFlg & SHM_EXEC ? true : false,
|
||||
};
|
||||
}
|
||||
|
||||
auto SyscallHandler::VMAFlags::fromFlags(int Flags) -> VMAFlags {
|
||||
return VMAFlags {
|
||||
.Shared = (Flags & MAP_SHARED) != 0, // also includes MAP_SHARED_VALIDATE
|
||||
};
|
||||
}
|
||||
|
||||
// SMC interactions
|
||||
bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState* Thread, int Signal, void* info, void* ucontext) {
|
||||
const auto FaultAddress = (uintptr_t)((siginfo_t*)info)->si_addr;
|
||||
@@ -57,7 +33,7 @@ bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState* Thread,
|
||||
auto VMATracking = &_SyscallHandler->VMATracking;
|
||||
|
||||
// If the write spans two pages, they will be flushed one at a time (generating two faults)
|
||||
auto Entry = VMATracking->LookupVMAUnsafe(FaultAddress);
|
||||
auto Entry = VMATracking->FindVMAEntry(FaultAddress);
|
||||
|
||||
// If an untracked address, or the mapping wasn't writable, it can't be handled here
|
||||
if (Entry == VMATracking->VMAs.end() || !Entry->second.Prot.Writable) {
|
||||
@@ -169,7 +145,7 @@ FEXCore::HLE::AOTIRCacheEntryLookupResult SyscallHandler::LookupAOTIRCacheEntry(
|
||||
// Get the first mapping after GuestAddr, or end
|
||||
// GuestAddr is inclusive
|
||||
// If the write spans two pages, they will be flushed one at a time (generating two faults)
|
||||
auto Entry = VMATracking.LookupVMAUnsafe(GuestAddr);
|
||||
auto Entry = VMATracking.FindVMAEntry(GuestAddr);
|
||||
if (Entry == VMATracking.VMAs.end()) {
|
||||
return {nullptr, 0};
|
||||
}
|
||||
@@ -177,12 +153,14 @@ FEXCore::HLE::AOTIRCacheEntryLookupResult SyscallHandler::LookupAOTIRCacheEntry(
|
||||
return {Entry->second.Resource ? Entry->second.Resource->AOTIRCacheEntry : nullptr, Entry->second.Base - Entry->second.Offset};
|
||||
}
|
||||
|
||||
// MMan Tracking
|
||||
void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int Prot, int Flags, int fd,
|
||||
off_t Offset) {
|
||||
Size = FEXCore::AlignUp(Size, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
void* SyscallHandler::GuestMmap(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags,
|
||||
int fd, off_t offset) {
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (length >> 32) == 0, "values must fit to 32 bits");
|
||||
|
||||
if (Flags & MAP_SHARED) {
|
||||
uint64_t Result {};
|
||||
size_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
if (flags & MAP_SHARED) {
|
||||
CTX->MarkMemoryShared(Thread);
|
||||
}
|
||||
|
||||
@@ -192,50 +170,33 @@ void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uintp
|
||||
// us to be more optimal by using GuardSignalDeferringSection instead
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
|
||||
|
||||
static uint64_t AnonSharedId = 1;
|
||||
|
||||
MappedResource* Resource = nullptr;
|
||||
|
||||
if (!(Flags & MAP_ANONYMOUS)) {
|
||||
struct stat64 buf;
|
||||
fstat64(fd, &buf);
|
||||
MRID mrid {buf.st_dev, buf.st_ino};
|
||||
|
||||
char Tmp[PATH_MAX];
|
||||
auto PathLength = FEX::get_fdpath(fd, Tmp);
|
||||
|
||||
if (PathLength != -1) {
|
||||
Tmp[PathLength] = '\0';
|
||||
auto [Iter, Inserted] = VMATracking.MappedResources.emplace(mrid, MappedResource {nullptr, nullptr, 0});
|
||||
Resource = &Iter->second;
|
||||
|
||||
if (Inserted) {
|
||||
Resource->AOTIRCacheEntry = CTX->LoadAOTIRCacheEntry(fextl::string(Tmp, PathLength));
|
||||
Resource->Iterator = Iter;
|
||||
}
|
||||
bool Map32Bit = !Is64Bit || (flags & FEX::HLE::X86_64_MAP_32BIT);
|
||||
if (Map32Bit) {
|
||||
Result = (uint64_t)Get32BitAllocator()->Mmap((void*)addr, length, prot, flags, fd, offset);
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
return reinterpret_cast<void*>(Result);
|
||||
}
|
||||
} else if (Flags & MAP_SHARED) {
|
||||
MRID mrid {SpecialDev::Anon, AnonSharedId++};
|
||||
|
||||
auto [Iter, Inserted] = VMATracking.MappedResources.emplace(mrid, MappedResource {nullptr, nullptr, 0});
|
||||
LOGMAN_THROW_A_FMT(Inserted == true, "VMA tracking error");
|
||||
Resource = &Iter->second;
|
||||
Resource->Iterator = Iter;
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (Result >> 32) == 0 || (Result >> 32) == 0xFFFFFFFF, "values must fit to 32 bits");
|
||||
} else {
|
||||
Resource = nullptr;
|
||||
Result = reinterpret_cast<uint64_t>(::mmap(reinterpret_cast<void*>(addr), length, prot, flags, fd, offset));
|
||||
if (Result == ~0ULL) {
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
}
|
||||
|
||||
VMATracking.SetUnsafe(CTX, Resource, Base, Offset, Size, VMAFlags::fromFlags(Flags), VMAProt::fromProt(Prot));
|
||||
FEX::HLE::_SyscallHandler->TrackMmap(Thread, Result, length, prot, flags, fd, offset);
|
||||
}
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
// VMATracking.Mutex can't be held while executing this, otherwise it hangs if the JIT is in the process of looking up code in the AOT JIT.
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, (uintptr_t)Base, Size);
|
||||
}
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessary(Thread, Result, Size);
|
||||
return reinterpret_cast<void*>(Result);
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size) {
|
||||
Size = FEXCore::AlignUp(Size, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
uint64_t SyscallHandler::GuestMunmap(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) {
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (reinterpret_cast<uintptr_t>(addr) >> 32) == 0, "values must fit to 32 bits: {}", fmt::ptr(addr));
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (length >> 32) == 0, "values must fit to 32 bits");
|
||||
|
||||
uint64_t Result {};
|
||||
uint64_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
{
|
||||
// Frontend calls this with nullptr Thread during initialization.
|
||||
@@ -243,121 +204,223 @@ void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
|
||||
|
||||
VMATracking.ClearUnsafe(CTX, Base, Size);
|
||||
if (reinterpret_cast<uintptr_t>(addr) < 0x1'0000'0000ULL) {
|
||||
Result = FEX::HLE::_SyscallHandler->Get32BitAllocator()->Munmap(addr, length);
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
return Result;
|
||||
}
|
||||
} else {
|
||||
Result = ::munmap(addr, length);
|
||||
if (Result == -1) {
|
||||
return -errno;
|
||||
}
|
||||
}
|
||||
FEX::HLE::_SyscallHandler->TrackMunmap(Thread, addr, length);
|
||||
}
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), Size);
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, (uintptr_t)Base, Size);
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMprotect(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int Prot) {
|
||||
Size = FEXCore::AlignUp(Size, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
uint64_t SyscallHandler::GuestMremap(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, void* old_address, size_t old_size,
|
||||
size_t new_size, int flags, void* new_address) {
|
||||
uint64_t Result {};
|
||||
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
VMATracking.ChangeUnsafe(Base, Size, VMAProt::fromProt(Prot));
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(FEX::HLE::_SyscallHandler->VMATracking.Mutex, Thread);
|
||||
if (Is64Bit) {
|
||||
Result = reinterpret_cast<uint64_t>(::mremap(old_address, old_size, new_size, flags, new_address));
|
||||
if (Result == -1) {
|
||||
return -errno;
|
||||
}
|
||||
} else {
|
||||
Result =
|
||||
reinterpret_cast<uint64_t>(FEX::HLE::_SyscallHandler->Get32BitAllocator()->Mremap(old_address, old_size, new_size, flags, new_address));
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
FEX::HLE::_SyscallHandler->TrackMremap(Thread, reinterpret_cast<uint64_t>(old_address), old_size, new_size, flags, Result);
|
||||
}
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, Base, Size);
|
||||
}
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessaryOnRemap(Thread, reinterpret_cast<uint64_t>(old_address), Result, old_size, new_size);
|
||||
return Result;
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMremap(FEXCore::Core::InternalThreadState* Thread, uintptr_t OldAddress, size_t OldSize, size_t NewSize,
|
||||
int flags, uintptr_t NewAddress) {
|
||||
uint64_t SyscallHandler::GuestMprotect(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t len, int prot) {
|
||||
uint64_t Result {};
|
||||
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(FEX::HLE::_SyscallHandler->VMATracking.Mutex, Thread);
|
||||
Result = ::mprotect(addr, len, prot);
|
||||
if (Result == -1) {
|
||||
return -errno;
|
||||
}
|
||||
|
||||
FEX::HLE::_SyscallHandler->TrackMprotect(Thread, addr, len, prot);
|
||||
}
|
||||
|
||||
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), len);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::GuestShmat(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, int shmid, const void* shmaddr, int shmflg) {
|
||||
auto CTX = Thread->CTX;
|
||||
uint64_t Result {};
|
||||
uint64_t Length {};
|
||||
CTX->MarkMemoryShared(Thread);
|
||||
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(FEX::HLE::_SyscallHandler->VMATracking.Mutex, Thread);
|
||||
if (Is64Bit) {
|
||||
Result = reinterpret_cast<uint64_t>(::shmat(shmid, shmaddr, shmflg));
|
||||
if (Result == -1) {
|
||||
return -errno;
|
||||
}
|
||||
} else {
|
||||
uint32_t Addr;
|
||||
Result = FEX::HLE::_SyscallHandler->Get32BitAllocator()->Shmat(shmid, shmaddr, shmflg, &Addr);
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
return Result;
|
||||
}
|
||||
Result = Addr;
|
||||
}
|
||||
|
||||
shmid_ds stat;
|
||||
|
||||
[[maybe_unused]] auto res = shmctl(shmid, IPC_STAT, &stat);
|
||||
LOGMAN_THROW_A_FMT(res != -1, "shmctl IPC_STAT failed");
|
||||
|
||||
Length = stat.shm_segsz;
|
||||
FEX::HLE::_SyscallHandler->TrackShmat(Thread, shmid, Result, shmflg, Length);
|
||||
}
|
||||
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessary(Thread, Result, Length);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::GuestShmdt(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, const void* shmaddr) {
|
||||
uint64_t Result {};
|
||||
uint64_t Length {};
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(FEX::HLE::_SyscallHandler->VMATracking.Mutex, Thread);
|
||||
if (Is64Bit) {
|
||||
Result = ::shmdt(shmaddr);
|
||||
if (Result == -1) {
|
||||
return -errno;
|
||||
}
|
||||
} else {
|
||||
Result = FEX::HLE::_SyscallHandler->Get32BitAllocator()->Shmdt(shmaddr);
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
Length = FEX::HLE::_SyscallHandler->TrackShmdt(Thread, reinterpret_cast<uintptr_t>(shmaddr));
|
||||
}
|
||||
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uintptr_t>(shmaddr), Length);
|
||||
return Result;
|
||||
}
|
||||
|
||||
// MMan Tracking
|
||||
void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
size_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
VMATracking::MappedResource* Resource = nullptr;
|
||||
|
||||
if (!(flags & MAP_ANONYMOUS)) {
|
||||
struct stat64 buf;
|
||||
fstat64(fd, &buf);
|
||||
VMATracking::MRID mrid {buf.st_dev, buf.st_ino};
|
||||
|
||||
char Tmp[PATH_MAX];
|
||||
auto PathLength = FEX::get_fdpath(fd, Tmp);
|
||||
|
||||
if (PathLength != -1) {
|
||||
Tmp[PathLength] = '\0';
|
||||
auto [Iter, Inserted] = VMATracking.EmplaceMappedResource(mrid, VMATracking::MappedResource {nullptr, nullptr, 0});
|
||||
Resource = &Iter->second;
|
||||
|
||||
if (Inserted) {
|
||||
Resource->AOTIRCacheEntry = CTX->LoadAOTIRCacheEntry(fextl::string(Tmp, PathLength));
|
||||
Resource->Iterator = Iter;
|
||||
}
|
||||
}
|
||||
} else if (flags & MAP_SHARED) {
|
||||
VMATracking::MRID mrid {VMATracking::SpecialDev::Anon, AnonSharedId++};
|
||||
|
||||
auto [Iter, Inserted] = VMATracking.EmplaceMappedResource(mrid, VMATracking::MappedResource {nullptr, nullptr, 0});
|
||||
LOGMAN_THROW_A_FMT(Inserted == true, "VMA tracking error");
|
||||
Resource = &Iter->second;
|
||||
Resource->Iterator = Iter;
|
||||
} else {
|
||||
Resource = nullptr;
|
||||
}
|
||||
|
||||
VMATracking.TrackVMARange(CTX, Resource, addr, offset, Size, VMATracking::VMAFlags::fromFlags(flags), VMATracking::VMAProt::fromProt(prot));
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length) {
|
||||
uint64_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
VMATracking.DeleteVMARange(CTX, reinterpret_cast<uintptr_t>(addr), Size);
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMprotect(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t len, int prot) {
|
||||
uint64_t Size = FEXCore::AlignUp(len, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
VMATracking.ChangeProtectionFlags(reinterpret_cast<uintptr_t>(addr), Size, VMATracking::VMAProt::fromProt(prot));
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMremap(FEXCore::Core::InternalThreadState* Thread, uint64_t OldAddress, size_t OldSize, size_t NewSize, int flags,
|
||||
uint64_t NewAddress) {
|
||||
OldSize = FEXCore::AlignUp(OldSize, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
NewSize = FEXCore::AlignUp(NewSize, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
const auto OldVMA = VMATracking.FindVMAEntry(OldAddress);
|
||||
|
||||
const auto OldVMA = VMATracking.LookupVMAUnsafe(OldAddress);
|
||||
const auto OldResource = OldVMA->second.Resource;
|
||||
const auto OldOffset = OldVMA->second.Offset + OldAddress - OldVMA->first;
|
||||
const auto OldFlags = OldVMA->second.Flags;
|
||||
const auto OldProt = OldVMA->second.Prot;
|
||||
|
||||
const auto OldResource = OldVMA->second.Resource;
|
||||
const auto OldOffset = OldVMA->second.Offset + OldAddress - OldVMA->first;
|
||||
const auto OldFlags = OldVMA->second.Flags;
|
||||
const auto OldProt = OldVMA->second.Prot;
|
||||
LOGMAN_THROW_A_FMT(OldVMA != VMATracking.VMAs.end(), "VMA Tracking corruption");
|
||||
|
||||
LOGMAN_THROW_A_FMT(OldVMA != VMATracking.VMAs.end(), "VMA Tracking corruption");
|
||||
if (OldSize == 0) {
|
||||
// Mirror existing mapping
|
||||
// must be a shared mapping
|
||||
LOGMAN_THROW_A_FMT(OldResource != nullptr, "VMA Tracking error");
|
||||
LOGMAN_THROW_A_FMT(OldFlags.Shared, "VMA Tracking error");
|
||||
VMATracking.TrackVMARange(CTX, OldResource, NewAddress, OldOffset, NewSize, OldFlags, OldProt);
|
||||
} else {
|
||||
|
||||
if (OldSize == 0) {
|
||||
// Mirror existing mapping
|
||||
// must be a shared mapping
|
||||
LOGMAN_THROW_A_FMT(OldResource != nullptr, "VMA Tracking error");
|
||||
LOGMAN_THROW_A_FMT(OldFlags.Shared, "VMA Tracking error");
|
||||
VMATracking.SetUnsafe(CTX, OldResource, NewAddress, OldOffset, NewSize, OldFlags, OldProt);
|
||||
} else {
|
||||
|
||||
// MREMAP_DONTUNMAP is kernel 5.7+
|
||||
#ifdef MREMAP_DONTUNMAP
|
||||
if (!(flags & MREMAP_DONTUNMAP))
|
||||
#ifndef MREMAP_DONTUNMAP
|
||||
// MREMAP_DONTUNMAP is kernel 5.7+ and might not exist
|
||||
#define MREMAP_DONTUNMAP 4
|
||||
#endif
|
||||
{
|
||||
VMATracking.ClearUnsafe(CTX, OldAddress, OldSize, OldResource);
|
||||
}
|
||||
|
||||
// Make anonymous mapping
|
||||
VMATracking.SetUnsafe(CTX, OldResource, NewAddress, OldOffset, NewSize, OldFlags, OldProt);
|
||||
if (!(flags & MREMAP_DONTUNMAP)) {
|
||||
VMATracking.DeleteVMARange(CTX, OldAddress, OldSize, OldResource);
|
||||
}
|
||||
}
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
if (OldAddress != NewAddress) {
|
||||
if (OldSize != 0) {
|
||||
// This also handles the MREMAP_DONTUNMAP case
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, OldAddress, OldSize);
|
||||
}
|
||||
} else {
|
||||
// If mapping shrunk, flush the unmapped region
|
||||
if (OldSize > NewSize) {
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, OldAddress + NewSize, OldSize - NewSize);
|
||||
}
|
||||
}
|
||||
// Make anonymous mapping
|
||||
VMATracking.TrackVMARange(CTX, OldResource, NewAddress, OldOffset, NewSize, OldFlags, OldProt);
|
||||
}
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState* Thread, int shmid, uintptr_t Base, int shmflg) {
|
||||
CTX->MarkMemoryShared(Thread);
|
||||
void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState* Thread, int shmid, uint64_t shmaddr, int shmflg, uint64_t Length) {
|
||||
VMATracking::MRID mrid {VMATracking::SpecialDev::SHM, static_cast<uint64_t>(shmid)};
|
||||
|
||||
shmid_ds stat;
|
||||
|
||||
[[maybe_unused]] auto res = shmctl(shmid, IPC_STAT, &stat);
|
||||
LOGMAN_THROW_A_FMT(res != -1, "shmctl IPC_STAT failed");
|
||||
|
||||
uint64_t Length = stat.shm_segsz;
|
||||
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
// TODO
|
||||
MRID mrid {SpecialDev::SHM, static_cast<uint64_t>(shmid)};
|
||||
|
||||
auto ResourceInserted = VMATracking.MappedResources.insert({mrid, {nullptr, nullptr, Length}});
|
||||
auto Resource = &ResourceInserted.first->second;
|
||||
if (ResourceInserted.second) {
|
||||
Resource->Iterator = ResourceInserted.first;
|
||||
}
|
||||
VMATracking.SetUnsafe(CTX, Resource, Base, 0, Length, VMAFlags::fromFlags(MAP_SHARED), VMAProt::fromSHM(shmflg));
|
||||
}
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, Base, Length);
|
||||
auto [Iter, Inserted] = VMATracking.EmplaceMappedResource(mrid, VMATracking::MappedResource {nullptr, nullptr, Length});
|
||||
auto Resource = &Iter->second;
|
||||
if (Inserted) {
|
||||
Resource->Iterator = Iter;
|
||||
}
|
||||
VMATracking.TrackVMARange(CTX, Resource, shmaddr, 0, Length, VMATracking::VMAFlags::fromFlags(MAP_SHARED), VMATracking::VMAProt::fromSHM(shmflg));
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base) {
|
||||
uintptr_t Length = 0;
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
Length = VMATracking.ClearShmUnsafe(CTX, Base);
|
||||
}
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
// This might over flush if the shm has holes in it
|
||||
_SyscallHandler->TM.InvalidateGuestCodeRange(Thread, Base, Length);
|
||||
}
|
||||
uint64_t SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState* Thread, uint64_t shmaddr) {
|
||||
return VMATracking.DeleteSHMRegion(CTX, reinterpret_cast<uintptr_t>(shmaddr));
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMadvise(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int advice) {
|
||||
|
||||
@@ -8,11 +8,34 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
#include <sys/shm.h>
|
||||
|
||||
namespace FEX::HLE::VMATracking {
|
||||
/// Helpers ///
|
||||
auto VMAProt::fromProt(int Prot) -> VMAProt {
|
||||
return VMAProt {
|
||||
.Readable = (Prot & PROT_READ) != 0,
|
||||
.Writable = (Prot & PROT_WRITE) != 0,
|
||||
.Executable = (Prot & PROT_EXEC) != 0,
|
||||
};
|
||||
}
|
||||
|
||||
auto VMAProt::fromSHM(int SHMFlg) -> VMAProt {
|
||||
return VMAProt {
|
||||
.Readable = true,
|
||||
.Writable = SHMFlg & SHM_RDONLY ? false : true,
|
||||
.Executable = SHMFlg & SHM_EXEC ? true : false,
|
||||
};
|
||||
}
|
||||
|
||||
auto VMAFlags::fromFlags(int Flags) -> VMAFlags {
|
||||
return VMAFlags {
|
||||
.Shared = (Flags & MAP_SHARED) != 0, // also includes MAP_SHARED_VALIDATE
|
||||
};
|
||||
}
|
||||
|
||||
namespace FEX::HLE {
|
||||
/// List Operations ///
|
||||
|
||||
inline void SyscallHandler::VMATracking::ListCheckVMALinks(VMAEntry* VMA) {
|
||||
inline void VMATracking::ListCheckVMALinks(VMAEntry* VMA) {
|
||||
if (VMA) {
|
||||
LOGMAN_THROW_A_FMT(VMA->ResourceNextVMA != VMA, "VMA tracking error");
|
||||
LOGMAN_THROW_A_FMT(VMA->ResourcePrevVMA != VMA, "VMA tracking error");
|
||||
@@ -21,7 +44,7 @@ inline void SyscallHandler::VMATracking::ListCheckVMALinks(VMAEntry* VMA) {
|
||||
|
||||
// Removes a VMA from corresponding MappedResource list
|
||||
// Returns true if list is empty
|
||||
bool SyscallHandler::VMATracking::ListRemove(VMAEntry* VMA) {
|
||||
bool VMATracking::ListRemove(VMAEntry* VMA) {
|
||||
LOGMAN_THROW_A_FMT(VMA->Resource != nullptr, "VMA tracking error");
|
||||
|
||||
// if it has prev, make prev to next
|
||||
@@ -55,7 +78,7 @@ bool SyscallHandler::VMATracking::ListRemove(VMAEntry* VMA) {
|
||||
|
||||
// Replaces a VMA in corresponding MappedResource list
|
||||
// Requires NewVMA->Resource, NewVMA->ResourcePrevVMA and NewVMA->ResourceNextVMA to be already setup
|
||||
void SyscallHandler::VMATracking::ListReplace(VMAEntry* VMA, VMAEntry* NewVMA) {
|
||||
void VMATracking::ListReplace(VMAEntry* VMA, VMAEntry* NewVMA) {
|
||||
LOGMAN_THROW_A_FMT(VMA->Resource != nullptr, "VMA tracking error");
|
||||
|
||||
LOGMAN_THROW_A_FMT(VMA->Resource == NewVMA->Resource, "VMA tracking error");
|
||||
@@ -84,7 +107,7 @@ void SyscallHandler::VMATracking::ListReplace(VMAEntry* VMA, VMAEntry* NewVMA) {
|
||||
|
||||
// Inserts a VMA in corresponding MappedResource list
|
||||
// Requires NewVMA->Resource, NewVMA->ResourcePrevVMA and NewVMA->ResourceNextVMA to be already setup
|
||||
void SyscallHandler::VMATracking::ListInsertAfter(VMAEntry* AfterVMA, VMAEntry* NewVMA) {
|
||||
void VMATracking::ListInsertAfter(VMAEntry* AfterVMA, VMAEntry* NewVMA) {
|
||||
LOGMAN_THROW_A_FMT(NewVMA->Resource != nullptr, "VMA tracking error");
|
||||
|
||||
LOGMAN_THROW_A_FMT(AfterVMA->Resource == NewVMA->Resource, "VMA tracking error");
|
||||
@@ -105,7 +128,7 @@ void SyscallHandler::VMATracking::ListInsertAfter(VMAEntry* AfterVMA, VMAEntry*
|
||||
|
||||
// Prepends a VMA
|
||||
// Requires NewVMA->Resource, NewVMA->ResourcePrevVMA and NewVMA->ResourceNextVMA to be already setup
|
||||
void SyscallHandler::VMATracking::ListPrepend(MappedResource* Resource, VMAEntry* NewVMA) {
|
||||
void VMATracking::ListPrepend(MappedResource* Resource, VMAEntry* NewVMA) {
|
||||
LOGMAN_THROW_A_FMT(Resource != nullptr, "VMA tracking error");
|
||||
|
||||
LOGMAN_THROW_A_FMT(NewVMA->Resource == Resource, "VMA tracking error");
|
||||
@@ -127,7 +150,7 @@ void SyscallHandler::VMATracking::ListPrepend(MappedResource* Resource, VMAEntry
|
||||
/// VMA tracking ///
|
||||
|
||||
// Lookup a VMA by address
|
||||
SyscallHandler::VMATracking::VMACIterator SyscallHandler::VMATracking::LookupVMAUnsafe(uint64_t GuestAddr) const {
|
||||
VMATracking::VMACIterator VMATracking::FindVMAEntry(uint64_t GuestAddr) const {
|
||||
auto Entry = VMAs.upper_bound(GuestAddr);
|
||||
|
||||
if (Entry != VMAs.begin()) {
|
||||
@@ -142,9 +165,11 @@ SyscallHandler::VMATracking::VMACIterator SyscallHandler::VMATracking::LookupVMA
|
||||
}
|
||||
|
||||
// Set or Replace mappings in a range with a new mapping
|
||||
void SyscallHandler::VMATracking::SetUnsafe(FEXCore::Context::Context* CTX, MappedResource* MappedResource, uintptr_t Base,
|
||||
uintptr_t Offset, uintptr_t Length, VMAFlags Flags, VMAProt Prot) {
|
||||
ClearUnsafe(CTX, Base, Length, MappedResource);
|
||||
void VMATracking::TrackVMARange(FEXCore::Context::Context* CTX, MappedResource* MappedResource, uintptr_t Base, uintptr_t Offset,
|
||||
uintptr_t Length, VMAFlags Flags, VMAProt Prot) {
|
||||
Mutex.check_lock_owned_by_self_as_write();
|
||||
|
||||
DeleteVMARange(CTX, Base, Length, MappedResource);
|
||||
|
||||
auto PrevResVMA = MappedResource ? MappedResource->FirstVMA : nullptr;
|
||||
auto NextResVMA = PrevResVMA ? PrevResVMA->ResourceNextVMA : nullptr;
|
||||
@@ -170,8 +195,9 @@ void SyscallHandler::VMATracking::SetUnsafe(FEXCore::Context::Context* CTX, Mapp
|
||||
|
||||
// Remove mappings in a range, possibly splitting them if needed and
|
||||
// freeing their associated MappedResource unless it is equal to PreservedMappedResource
|
||||
void SyscallHandler::VMATracking::ClearUnsafe(FEXCore::Context::Context* CTX, uintptr_t Base, uintptr_t Length,
|
||||
MappedResource* PreservedMappedResource) {
|
||||
void VMATracking::DeleteVMARange(FEXCore::Context::Context* CTX, uintptr_t Base, uintptr_t Length, MappedResource* PreservedMappedResource) {
|
||||
Mutex.check_lock_owned_by_self_as_write();
|
||||
|
||||
const auto Top = Base + Length;
|
||||
|
||||
// find the first Mapping at or after the Range ends, or ::end()
|
||||
@@ -255,7 +281,9 @@ void SyscallHandler::VMATracking::ClearUnsafe(FEXCore::Context::Context* CTX, ui
|
||||
}
|
||||
|
||||
// Change flags of mappings in a range and split the mappings if needed
|
||||
void SyscallHandler::VMATracking::ChangeUnsafe(uintptr_t Base, uintptr_t Length, VMAProt NewProt) {
|
||||
void VMATracking::ChangeProtectionFlags(uintptr_t Base, uintptr_t Length, VMAProt NewProt) {
|
||||
Mutex.check_lock_owned_by_self_as_write();
|
||||
|
||||
// This needs to handle multiple split-merge strategies:
|
||||
// 1) Exact overlap - No Split, no Merge. Only protection tracking changes.
|
||||
// 2) Exact base overlap - Single insert, can never fail.
|
||||
@@ -485,7 +513,7 @@ void SyscallHandler::VMATracking::ChangeUnsafe(uintptr_t Base, uintptr_t Length,
|
||||
}
|
||||
|
||||
// This matches the peculiarities algorithm used in linux ksys_shmdt (linux kernel 5.16, ipc/shm.c)
|
||||
uintptr_t SyscallHandler::VMATracking::ClearShmUnsafe(FEXCore::Context::Context* CTX, uintptr_t Base) {
|
||||
uintptr_t VMATracking::DeleteSHMRegion(FEXCore::Context::Context* CTX, uintptr_t Base) {
|
||||
|
||||
// Find first VMA at or after Base
|
||||
// Iterate until first SHM VMA, with matching offset, get length
|
||||
@@ -525,4 +553,4 @@ uintptr_t SyscallHandler::VMATracking::ClearShmUnsafe(FEXCore::Context::Context*
|
||||
|
||||
return ShmLength;
|
||||
}
|
||||
} // namespace FEX::HLE
|
||||
} // namespace FEX::HLE::VMATracking
|
||||
@@ -0,0 +1,134 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct AOTIRCacheEntry;
|
||||
}
|
||||
|
||||
namespace FEX::HLE::VMATracking {
|
||||
///// VMA (Virtual Memory Area) tracking /////
|
||||
|
||||
namespace SpecialDev {
|
||||
static constexpr uint64_t Anon = 0x1'0000'0000; // Anonymous shared mapping, id is incrementing allocation number
|
||||
static constexpr uint64_t SHM = 0x2'0000'0000; // sys-v shm, id is shmid
|
||||
}; // namespace SpecialDev
|
||||
|
||||
// Memory Resource ID
|
||||
// An id that can be used to identify when shared mappings actually have the same backing storage
|
||||
// when dev != SpecialDev::Anon, this is unique system wide
|
||||
struct MRID {
|
||||
uint64_t dev; // kernel dev_t is actually 32-bits, we use the extra bits to track SpecialDevs
|
||||
uint64_t id;
|
||||
|
||||
bool operator<(const MRID& other) const {
|
||||
return std::tie(dev, id) < std::tie(other.dev, other.id);
|
||||
}
|
||||
};
|
||||
|
||||
struct VMAEntry;
|
||||
|
||||
// Used to all MAP_SHARED VMAs of a system resource.
|
||||
struct MappedResource {
|
||||
using ContainerType = fextl::map<MRID, MappedResource>;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry* AOTIRCacheEntry;
|
||||
// Pointer to lowest memory range this file is mapped to
|
||||
VMAEntry* FirstVMA;
|
||||
uint64_t Length; // 0 if not fixed size
|
||||
ContainerType::iterator Iterator;
|
||||
};
|
||||
|
||||
union VMAProt {
|
||||
struct {
|
||||
bool Readable : 1;
|
||||
bool Writable : 1;
|
||||
bool Executable : 1;
|
||||
};
|
||||
uint8_t All : 3;
|
||||
|
||||
static VMAProt fromProt(int Prot);
|
||||
static VMAProt fromSHM(int SHMFlg);
|
||||
};
|
||||
|
||||
struct VMAFlags {
|
||||
bool Shared : 1;
|
||||
|
||||
static VMAFlags fromFlags(int Flags);
|
||||
};
|
||||
|
||||
struct VMAEntry {
|
||||
MappedResource* Resource;
|
||||
|
||||
// these are for intrusive linked list tracking, starting from Resource->FirstVMA and ordered by address
|
||||
VMAEntry* ResourcePrevVMA;
|
||||
VMAEntry* ResourceNextVMA;
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Offset;
|
||||
uint64_t Length;
|
||||
|
||||
VMAFlags Flags;
|
||||
VMAProt Prot;
|
||||
};
|
||||
|
||||
struct VMATracking {
|
||||
// Held while reading/writing this struct
|
||||
FEXCore::ForkableSharedMutex Mutex;
|
||||
|
||||
// Memory ranges indexed by page aligned starting address
|
||||
fextl::map<uint64_t, VMAEntry> VMAs;
|
||||
|
||||
using VMACIterator = decltype(VMAs)::const_iterator;
|
||||
|
||||
// Find a VMA entry associated with the memory address.
|
||||
// Used by `mremap`, and SIGSEGV handler to find previously mapped ranges, and `AOTIR` cache to find cache entries.
|
||||
// - Mutex must be at least shared_locked before calling
|
||||
VMACIterator FindVMAEntry(uint64_t GuestAddr) const;
|
||||
|
||||
// Adds a new VMA Range to be tracked, along with a `MappedResource` associated with that VMA range.
|
||||
// Primarily matches `mmap` semantics, but also used by `mremap`, and `shmat`, as they all can add new VMA ranges to be tracked.
|
||||
// - Mutex must be unique_locked before calling
|
||||
void TrackVMARange(FEXCore::Context::Context* Ctx, MappedResource* MappedResource, uintptr_t Base, uintptr_t Offset, uintptr_t Length,
|
||||
VMAFlags Flags, VMAProt Prot);
|
||||
|
||||
// Deletes a VMA range provided from tracking.
|
||||
// Matches `munmap` semantics, and `mremap` with `MREMAP_DONTUNMAP` flag set.
|
||||
// Deletes internal `MappedResource` that correlates with the range **unless** it matches `PreservedMappedResource`
|
||||
// - Mutex must be unique_locked before calling
|
||||
void DeleteVMARange(FEXCore::Context::Context* Ctx, uintptr_t Base, uintptr_t Length, MappedResource* PreservedMappedResource = nullptr);
|
||||
|
||||
// Changes the protections tracking for the VMA range provided.
|
||||
// Matches `mprotect` semantics.
|
||||
// - Mutex must be unique_locked before calling
|
||||
void ChangeProtectionFlags(uintptr_t Base, uintptr_t Length, VMAProt Prot);
|
||||
|
||||
// Deletes the SHM region mapped at Base from tracking.
|
||||
// Matches `shmdt` semantics.
|
||||
// - Mutex must be unique_locked before calling
|
||||
// Returns the Size of the Shm or 0 if not found
|
||||
uintptr_t DeleteSHMRegion(FEXCore::Context::Context* Ctx, uintptr_t Base);
|
||||
|
||||
// Emplaces a new `MappedResource` to track.
|
||||
// Used for `mmap` and `shmat` resources; Anonymous, FD, and SHM depending on flags.
|
||||
template<class... Args>
|
||||
inline auto EmplaceMappedResource(Args&&... args) {
|
||||
return MappedResources.emplace(args...);
|
||||
}
|
||||
private:
|
||||
bool ListRemove(VMAEntry* Mapping);
|
||||
void ListReplace(VMAEntry* Mapping, VMAEntry* NewMapping);
|
||||
void ListInsertAfter(VMAEntry* Mapping, VMAEntry* NewMapping);
|
||||
void ListPrepend(MappedResource* Resource, VMAEntry* NewVMA);
|
||||
static void ListCheckVMALinks(VMAEntry* VMA);
|
||||
|
||||
MappedResource::ContainerType MappedResources;
|
||||
};
|
||||
|
||||
|
||||
} // namespace FEX::HLE::VMATracking
|
||||
@@ -297,7 +297,7 @@ void ThreadManager::Step() {
|
||||
// Walk the threads and tell them to clear their caches
|
||||
// Useful when our block size is set to a large number and we need to step a single instruction
|
||||
for (auto& Thread : Threads) {
|
||||
CTX->ClearCodeCache(Thread->Thread);
|
||||
CTX->ClearCodeCache(Thread->Thread, false);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -5,8 +5,6 @@
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <csetjmp>
|
||||
|
||||
namespace FEX::LinuxEmulation::Threads {
|
||||
void* StackTracker::AllocateStackObject() {
|
||||
std::lock_guard lk {DeadStackPoolMutex};
|
||||
@@ -190,6 +188,143 @@ __attribute__((naked)) void StackPivotAndCall(void* Arg, FEXCore::Threads::Threa
|
||||
}
|
||||
#endif
|
||||
namespace PThreads {
|
||||
namespace LongJump {
|
||||
// This is a custom long jump implementation that avoids the glibc implementation.
|
||||
// This is required behaviour because glibc's fortification checks don't understand stack pivots.
|
||||
// FEX requires a stack pivot to work through a long jump, so these two features are at odds with each other.
|
||||
#ifdef _M_ARM_64
|
||||
struct JumpBuf {
|
||||
// All the registers that are required by AAPCS64 to save.
|
||||
// GPRs
|
||||
// X19, X20, X21, X22,
|
||||
// X23, X24, X25, X26,
|
||||
// X27, X28, X29, X30,
|
||||
//
|
||||
// Lower 64-bits:
|
||||
// V8, V9, V10, V11,
|
||||
// V12, V13, V14, V15,
|
||||
//
|
||||
// SP,
|
||||
uint64_t Registers[21];
|
||||
};
|
||||
FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
__asm volatile(R"(
|
||||
// x0 contains the jumpbuffer
|
||||
stp x19, x20, [x0, #( 0 * 8)];
|
||||
stp x21, x22, [x0, #( 2 * 8)];
|
||||
stp x23, x24, [x0, #( 4 * 8)];
|
||||
stp x25, x26, [x0, #( 6 * 8)];
|
||||
stp x27, x28, [x0, #( 8 * 8)];
|
||||
stp x29, x30, [x0, #(10 * 8)];
|
||||
|
||||
// FPRs
|
||||
stp d8, d9, [x0, #(12 * 8)];
|
||||
stp d10, d11, [x0, #(14 * 8)];
|
||||
stp d12, d13, [x0, #(16 * 8)];
|
||||
stp d14, d15, [x0, #(18 * 8)];
|
||||
|
||||
// Move SP in to a temporary to store.
|
||||
mov x1, sp;
|
||||
str x1, [x0, #(19 * 8)];
|
||||
|
||||
// Return zero to signify this is the SetJump.
|
||||
mov x0, #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value) {
|
||||
__asm volatile(R"(
|
||||
// x0 contains the jumpbuffer
|
||||
ldp x19, x20, [x0, #( 0 * 8)];
|
||||
ldp x21, x22, [x0, #( 2 * 8)];
|
||||
ldp x23, x24, [x0, #( 4 * 8)];
|
||||
ldp x25, x26, [x0, #( 6 * 8)];
|
||||
ldp x27, x28, [x0, #( 8 * 8)];
|
||||
ldp x29, x30, [x0, #(10 * 8)];
|
||||
|
||||
// FPRs
|
||||
ldp d8, d9, [x0, #(12 * 8)];
|
||||
ldp d10, d11, [x0, #(14 * 8)];
|
||||
ldp d12, d13, [x0, #(16 * 8)];
|
||||
ldp d14, d15, [x0, #(18 * 8)];
|
||||
|
||||
// Load SP in to temporary then move
|
||||
ldr x0, [x0, #(19 * 8)];
|
||||
mov sp, x0;
|
||||
|
||||
// Move value in to result register
|
||||
mov x0, x1;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
#else
|
||||
struct JumpBuf {
|
||||
// Registers to preserve
|
||||
// RBX, RSP, RBP, R12, R13, R14, R15,
|
||||
// <return address>
|
||||
uint64_t Registers[8];
|
||||
};
|
||||
|
||||
__attribute__((naked)) uint64_t SetJump(JumpBuf& Buffer) {
|
||||
__asm volatile(R"(
|
||||
.intel_syntax noprefix;
|
||||
// rdi contains the jumpbuffer
|
||||
mov [rdi + (0 * 8)], rbx;
|
||||
mov [rdi + (1 * 8)], rsp;
|
||||
mov [rdi + (2 * 8)], rbp;
|
||||
mov [rdi + (3 * 8)], r12;
|
||||
mov [rdi + (4 * 8)], r13;
|
||||
mov [rdi + (5 * 8)], r14;
|
||||
mov [rdi + (6 * 8)], r15;
|
||||
|
||||
// Return address is on the stack, load it and store
|
||||
mov rsi, [rsp];
|
||||
mov [rdi + (7 * 8)], rsi;
|
||||
|
||||
// Return zero to signify this is the SetJump.
|
||||
mov rax, 0;
|
||||
ret;
|
||||
|
||||
.att_syntax prefix;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
__attribute__((naked)) void LongJump(JumpBuf& Buffer, uint64_t Value) {
|
||||
__asm volatile(R"(
|
||||
.intel_syntax noprefix;
|
||||
// rdi contains the jumpbuffer
|
||||
mov rbx, [rdi + (0 * 8)];
|
||||
mov rsp, [rdi + (1 * 8)];
|
||||
mov rbp, [rdi + (2 * 8)];
|
||||
mov r12, [rdi + (3 * 8)];
|
||||
mov r13, [rdi + (4 * 8)];
|
||||
mov r14, [rdi + (5 * 8)];
|
||||
mov r15, [rdi + (6 * 8)];
|
||||
|
||||
// Move value in to result register
|
||||
mov rax, rsi;
|
||||
|
||||
// Pop the dead return address off the stack
|
||||
pop rsi;
|
||||
|
||||
// Load the original return address from the jumpbuffer
|
||||
mov rsi, [rdi + (7 * 8)];
|
||||
|
||||
// Return using a jump
|
||||
jmp rsi;
|
||||
|
||||
.att_syntax prefix;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
#endif
|
||||
}; // namespace LongJump
|
||||
void* InitializeThread(void* Ptr);
|
||||
|
||||
class PThread final : public FEXCore::Threads::Thread {
|
||||
@@ -260,7 +395,7 @@ namespace PThreads {
|
||||
return STracker;
|
||||
}
|
||||
|
||||
void SetupLongJump(std::jmp_buf* exit_resolver) {
|
||||
void SetupLongJump(LongJump::JumpBuf* exit_resolver) {
|
||||
_exit_resolver = exit_resolver;
|
||||
}
|
||||
|
||||
@@ -268,7 +403,8 @@ namespace PThreads {
|
||||
void LongJumpExit(FEX::HLE::ThreadStateObject* ThreadObject, uint32_t Status) {
|
||||
this->Status = Status;
|
||||
this->ThreadObject = ThreadObject;
|
||||
std::longjmp(*_exit_resolver, 1);
|
||||
LongJump::LongJump(*_exit_resolver, 1);
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
uint32_t GetStatus() const {
|
||||
@@ -286,7 +422,7 @@ namespace PThreads {
|
||||
void* UserArg;
|
||||
void* Stack {};
|
||||
|
||||
std::jmp_buf* _exit_resolver {};
|
||||
LongJump::JumpBuf* _exit_resolver {};
|
||||
FEX::HLE::ThreadStateObject* ThreadObject {};
|
||||
uint32_t Status {};
|
||||
};
|
||||
@@ -297,11 +433,11 @@ namespace PThreads {
|
||||
PThread* Thread {reinterpret_cast<PThread*>(Ptr)};
|
||||
StackBase = Thread->GetPivotStack();
|
||||
STracker = Thread->GetStackTracker();
|
||||
std::jmp_buf exit_resolver {};
|
||||
LongJump::JumpBuf exit_resolver {};
|
||||
|
||||
bool LongJumpExit {};
|
||||
|
||||
if (setjmp(exit_resolver) == 0) {
|
||||
if (LongJump::SetJump(exit_resolver) == 0) {
|
||||
Thread->SetupLongJump(&exit_resolver);
|
||||
// Run the user function.
|
||||
// `Thread` object is dead after this function returns.
|
||||
|
||||
@@ -11,47 +11,18 @@ $end_info$
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/shm.h>
|
||||
#include <system_error>
|
||||
#include <filesystem>
|
||||
|
||||
namespace FEX::HLE::x32 {
|
||||
|
||||
void* x32SyscallHandler::GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
LOGMAN_THROW_A_FMT((length >> 32) == 0, "values must fit to 32 bits");
|
||||
|
||||
auto Result = (uint64_t)GetAllocator()->Mmap((void*)addr, length, prot, flags, fd, offset);
|
||||
|
||||
LOGMAN_THROW_A_FMT((Result >> 32) == 0 || (Result >> 32) == 0xFFFFFFFF, "values must fit to 32 bits");
|
||||
|
||||
if (!FEX::HLE::HasSyscallError(Result)) {
|
||||
FEX::HLE::_SyscallHandler->TrackMmap(Thread, Result, length, prot, flags, fd, offset);
|
||||
return (void*)Result;
|
||||
} else {
|
||||
errno = -Result;
|
||||
return MAP_FAILED;
|
||||
}
|
||||
}
|
||||
|
||||
int x32SyscallHandler::GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) {
|
||||
LOGMAN_THROW_A_FMT((uintptr_t(addr) >> 32) == 0, "values must fit to 32 bits");
|
||||
LOGMAN_THROW_A_FMT((length >> 32) == 0, "values must fit to 32 bits");
|
||||
|
||||
auto Result = GetAllocator()->Munmap(addr, length);
|
||||
|
||||
if (Result == 0) {
|
||||
FEX::HLE::_SyscallHandler->TrackMunmap(Thread, (uintptr_t)addr, length);
|
||||
return Result;
|
||||
} else {
|
||||
errno = -Result;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
void RegisterMemory(FEX::HLE::SyscallHandler* Handler) {
|
||||
struct old_mmap_struct {
|
||||
uint32_t addr;
|
||||
@@ -62,46 +33,27 @@ void RegisterMemory(FEX::HLE::SyscallHandler* Handler) {
|
||||
uint32_t offset;
|
||||
};
|
||||
REGISTER_SYSCALL_IMPL_X32(mmap, [](FEXCore::Core::CpuStateFrame* Frame, const old_mmap_struct* arg) -> uint64_t {
|
||||
uint64_t Result = (uint64_t) static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)
|
||||
->GuestMmap(Frame->Thread, reinterpret_cast<void*>(arg->addr), arg->len, arg->prot, arg->flags, arg->fd, arg->offset);
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
return reinterpret_cast<uint64_t>(FEX::HLE::_SyscallHandler->GuestMmap(false, Frame->Thread, reinterpret_cast<void*>(arg->addr),
|
||||
arg->len, arg->prot, arg->flags, arg->fd, arg->offset));
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(
|
||||
mmap2, [](FEXCore::Core::CpuStateFrame* Frame, uint32_t addr, uint32_t length, int prot, int flags, int fd, uint32_t pgoffset) -> uint64_t {
|
||||
uint64_t Result = (uint64_t) static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)
|
||||
->GuestMmap(Frame->Thread, reinterpret_cast<void*>(addr), length, prot, flags, fd, (uint64_t)pgoffset * 0x1000);
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
return reinterpret_cast<uint64_t>(FEX::HLE::_SyscallHandler->GuestMmap(false, Frame->Thread, reinterpret_cast<void*>(addr), length,
|
||||
prot, flags, fd, (uint64_t)pgoffset * 0x1000));
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(munmap, [](FEXCore::Core::CpuStateFrame* Frame, void* addr, size_t length) -> uint64_t {
|
||||
uint64_t Result =
|
||||
(uint64_t) static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GuestMunmap(Frame->Thread, addr, length);
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
return FEX::HLE::_SyscallHandler->GuestMunmap(Frame->Thread, addr, length);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mprotect, [](FEXCore::Core::CpuStateFrame* Frame, void* addr, uint32_t len, int prot) -> uint64_t {
|
||||
uint64_t Result = ::mprotect(addr, len, prot);
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackMprotect(Frame->Thread, (uintptr_t)addr, len, prot);
|
||||
}
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
return FEX::HLE::_SyscallHandler->GuestMprotect(Frame->Thread, addr, len, prot);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(
|
||||
mremap, [](FEXCore::Core::CpuStateFrame* Frame, void* old_address, size_t old_size, size_t new_size, int flags, void* new_address) -> uint64_t {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(
|
||||
static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->Mremap(old_address, old_size, new_size, flags, new_address));
|
||||
|
||||
if (!FEX::HLE::HasSyscallError(Result)) {
|
||||
FEX::HLE::_SyscallHandler->TrackMremap(Frame->Thread, (uintptr_t)old_address, old_size, new_size, flags, Result);
|
||||
}
|
||||
|
||||
return Result;
|
||||
return FEX::HLE::_SyscallHandler->GuestMremap(false, Frame->Thread, old_address, old_size, new_size, flags, new_address);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mlockall, [](FEXCore::Core::CpuStateFrame* Frame, int flags) -> uint64_t {
|
||||
@@ -115,29 +67,11 @@ void RegisterMemory(FEX::HLE::SyscallHandler* Handler) {
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(shmat, [](FEXCore::Core::CpuStateFrame* Frame, int shmid, const void* shmaddr, int shmflg) -> uint64_t {
|
||||
// also implemented in ipc:OP_SHMAT
|
||||
uint32_t ResultAddr {};
|
||||
uint64_t Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)
|
||||
->GetAllocator()
|
||||
->Shmat(shmid, reinterpret_cast<const void*>(shmaddr), shmflg, &ResultAddr);
|
||||
|
||||
if (!FEX::HLE::HasSyscallError(Result)) {
|
||||
FEX::HLE::_SyscallHandler->TrackShmat(Frame->Thread, shmid, ResultAddr, shmflg);
|
||||
return ResultAddr;
|
||||
} else {
|
||||
return Result;
|
||||
}
|
||||
return FEX::HLE::_SyscallHandler->GuestShmat(false, Frame->Thread, shmid, shmaddr, shmflg);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(shmdt, [](FEXCore::Core::CpuStateFrame* Frame, const void* shmaddr) -> uint64_t {
|
||||
// also implemented in ipc:OP_SHMDT
|
||||
uint64_t Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->Shmdt(shmaddr);
|
||||
|
||||
if (!FEX::HLE::HasSyscallError(Result)) {
|
||||
FEX::HLE::_SyscallHandler->TrackShmdt(Frame->Thread, (uintptr_t)shmaddr);
|
||||
}
|
||||
|
||||
return Result;
|
||||
return FEX::HLE::_SyscallHandler->GuestShmdt(false, Frame->Thread, shmaddr);
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -293,22 +293,14 @@ uint64_t _ipc(FEXCore::Core::CpuStateFrame* Frame, uint32_t call, uint32_t first
|
||||
// shmat explicitly doesn't support version 1.
|
||||
return -EINVAL;
|
||||
}
|
||||
// also implemented in memory:shmat
|
||||
Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)
|
||||
->GetAllocator()
|
||||
->Shmat(first, reinterpret_cast<const void*>(ptr), second, reinterpret_cast<uint32_t*>(third));
|
||||
auto Result = FEX::HLE::_SyscallHandler->GuestShmat(false, Frame->Thread, first, reinterpret_cast<const void*>(ptr), second);
|
||||
if (!FEX::HLE::HasSyscallError(Result)) {
|
||||
FEX::HLE::_SyscallHandler->TrackShmat(Frame->Thread, first, *reinterpret_cast<uint32_t*>(third), second);
|
||||
*reinterpret_cast<uint32_t*>(third) = Result;
|
||||
}
|
||||
break;
|
||||
return Result;
|
||||
}
|
||||
case OP_SHMDT: {
|
||||
// also implemented in memory:shmdt
|
||||
Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->Shmdt(reinterpret_cast<void*>(ptr));
|
||||
if (!FEX::HLE::HasSyscallError(Result)) {
|
||||
FEX::HLE::_SyscallHandler->TrackShmdt(Frame->Thread, ptr);
|
||||
}
|
||||
break;
|
||||
return FEX::HLE::_SyscallHandler->GuestShmdt(false, Frame->Thread, reinterpret_cast<const void*>(ptr));
|
||||
}
|
||||
case OP_SHMGET: {
|
||||
Result = ::shmget(first, second, third);
|
||||
|
||||
@@ -41,8 +41,16 @@ public:
|
||||
FEX::HLE::MemAllocator* GetAllocator() {
|
||||
return AllocHandler.get();
|
||||
}
|
||||
void* GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) override;
|
||||
int GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) override;
|
||||
FEX::HLE::MemAllocator* Get32BitAllocator() override {
|
||||
return GetAllocator();
|
||||
}
|
||||
|
||||
void* GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) override {
|
||||
return FEX::HLE::SyscallHandler::GuestMmap(false, Thread, addr, length, prot, flags, fd, offset);
|
||||
}
|
||||
uint64_t GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) override {
|
||||
return FEX::HLE::SyscallHandler::GuestMunmap(false, Thread, addr, length);
|
||||
}
|
||||
|
||||
void RegisterSyscall_32(int SyscallNumber, int32_t HostSyscallNumber, FEXCore::IR::SyscallFlags Flags,
|
||||
#ifdef DEBUG_STRACE
|
||||
|
||||
@@ -20,109 +20,43 @@ $end_info$
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEX::HLE::x64 {
|
||||
|
||||
void* x64SyscallHandler::GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
uint64_t Result {};
|
||||
|
||||
bool Map32Bit = flags & FEX::HLE::X86_64_MAP_32BIT;
|
||||
if (Map32Bit) {
|
||||
Result = (uint64_t)Get32BitAllocator()->Mmap(addr, length, prot, flags, fd, offset);
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
errno = -Result;
|
||||
Result = -1;
|
||||
}
|
||||
} else {
|
||||
Result = reinterpret_cast<uint64_t>(::mmap(reinterpret_cast<void*>(addr), length, prot, flags, fd, offset));
|
||||
}
|
||||
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackMmap(Thread, (uintptr_t)Result, length, prot, flags, fd, offset);
|
||||
}
|
||||
|
||||
return reinterpret_cast<void*>(Result);
|
||||
}
|
||||
|
||||
int x64SyscallHandler::GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) {
|
||||
uint64_t Result {};
|
||||
if (reinterpret_cast<uintptr_t>(addr) < 0x1'0000'0000ULL) {
|
||||
Result = Get32BitAllocator()->Munmap(addr, length);
|
||||
|
||||
if (FEX::HLE::HasSyscallError(Result)) {
|
||||
errno = -Result;
|
||||
Result = -1;
|
||||
}
|
||||
} else {
|
||||
Result = ::munmap(addr, length);
|
||||
}
|
||||
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackMunmap(Thread, reinterpret_cast<uintptr_t>(addr), length);
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
void RegisterMemory(FEX::HLE::SyscallHandler* Handler) {
|
||||
using namespace FEXCore::IR;
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(
|
||||
mmap, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, void* addr, size_t length, int prot, int flags, int fd, off_t offset) -> uint64_t {
|
||||
uint64_t Result = (uint64_t) static_cast<FEX::HLE::x64::x64SyscallHandler*>(FEX::HLE::_SyscallHandler)
|
||||
->GuestMmap(Frame->Thread, addr, length, prot, flags, fd, offset);
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
return (uint64_t)FEX::HLE::_SyscallHandler->GuestMmap(Frame->Thread, addr, length, prot, flags, fd, offset);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(
|
||||
munmap, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, void* addr, size_t length) -> uint64_t {
|
||||
uint64_t Result = static_cast<FEX::HLE::x64::x64SyscallHandler*>(FEX::HLE::_SyscallHandler)->GuestMunmap(Frame->Thread, addr, length);
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(munmap, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, void* addr, size_t length) -> uint64_t {
|
||||
return FEX::HLE::_SyscallHandler->GuestMunmap(Frame->Thread, addr, length);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(
|
||||
mremap, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, void* old_address, size_t old_size, size_t new_size, int flags, void* new_address) -> uint64_t {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(::mremap(old_address, old_size, new_size, flags, new_address));
|
||||
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackMremap(Frame->Thread, (uintptr_t)old_address, old_size, new_size, flags, Result);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
return FEX::HLE::_SyscallHandler->GuestMremap(true, Frame->Thread, old_address, old_size, new_size, flags, new_address);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(mprotect, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, void* addr, size_t len, int prot) -> uint64_t {
|
||||
uint64_t Result = ::mprotect(addr, len, prot);
|
||||
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackMprotect(Frame->Thread, (uintptr_t)addr, len, prot);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
return FEX::HLE::_SyscallHandler->GuestMprotect(Frame->Thread, addr, len, prot);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(shmat, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int shmid, const void* shmaddr, int shmflg) -> uint64_t {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(shmat(shmid, shmaddr, shmflg));
|
||||
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackShmat(Frame->Thread, shmid, Result, shmflg);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
([](FEXCore::Core::CpuStateFrame* Frame, int shmid, const void* shmaddr, int shmflg) -> uint64_t {
|
||||
return FEX::HLE::_SyscallHandler->GuestShmat(true, Frame->Thread, shmid, shmaddr, shmflg);
|
||||
}));
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64_FLAGS(shmdt, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, const void* shmaddr) -> uint64_t {
|
||||
uint64_t Result = ::shmdt(shmaddr);
|
||||
|
||||
if (Result != -1) {
|
||||
FEX::HLE::_SyscallHandler->TrackShmdt(Frame->Thread, (uintptr_t)shmaddr);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
return FEX::HLE::_SyscallHandler->GuestShmdt(true, Frame->Thread, shmaddr);
|
||||
});
|
||||
}
|
||||
} // namespace FEX::HLE::x64
|
||||
@@ -35,8 +35,12 @@ class x64SyscallHandler final : public FEX::HLE::SyscallHandler {
|
||||
public:
|
||||
x64SyscallHandler(FEXCore::Context::Context* ctx, FEX::HLE::SignalDelegator* _SignalDelegation, FEX::HLE::ThunkHandler* ThunkHandler);
|
||||
|
||||
void* GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) override;
|
||||
int GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) override;
|
||||
void* GuestMmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags, int fd, off_t offset) override {
|
||||
return FEX::HLE::SyscallHandler::GuestMmap(true, Thread, addr, length, prot, flags, fd, offset);
|
||||
}
|
||||
uint64_t GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) override {
|
||||
return FEX::HLE::SyscallHandler::GuestMunmap(true, Thread, addr, length);
|
||||
}
|
||||
|
||||
|
||||
void RegisterSyscall_64(int SyscallNumber, int32_t HostSyscallNumber, FEXCore::IR::SyscallFlags Flags,
|
||||
|
||||
@@ -715,8 +715,14 @@ void LoadUnique32BitSigreturn(VDSOMapping* Mapping, FEX::HLE::SyscallHandler* co
|
||||
memcpy(reinterpret_cast<void*>(VDSOPointers.VDSO_kernel_rt_sigreturn), &rt_sigreturn_32_code.at(0), rt_sigreturn_32_code.size());
|
||||
|
||||
mprotect(Mapping->OptionalSigReturnMapping, Mapping->OptionalMappingSize, PROT_READ | PROT_EXEC);
|
||||
Handler->TrackMmap(nullptr, reinterpret_cast<uintptr_t>(Mapping->OptionalSigReturnMapping), Mapping->OptionalMappingSize,
|
||||
PROT_READ | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(Handler->VMATracking.Mutex, nullptr);
|
||||
FEX::HLE::_SyscallHandler->TrackMmap(nullptr, reinterpret_cast<uint64_t>(Mapping->OptionalSigReturnMapping),
|
||||
Mapping->OptionalMappingSize, PROT_READ | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
|
||||
}
|
||||
|
||||
FEX::HLE::_SyscallHandler->InvalidateCodeRangeIfNecessary(nullptr, reinterpret_cast<uint64_t>(Mapping->OptionalSigReturnMapping),
|
||||
Mapping->OptionalMappingSize);
|
||||
}
|
||||
|
||||
void UnloadVDSOMapping(const VDSOMapping& Mapping) {
|
||||
|
||||
@@ -332,7 +332,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (!CTX->InitCore()) {
|
||||
return 1;
|
||||
}
|
||||
auto ParentThread = SyscallHandler->TM.CreateThread(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
auto ParentThread = SyscallHandler->TM.CreateThread(Loader.DefaultRIP(), 0);
|
||||
SyscallHandler->TM.TrackThread(ParentThread);
|
||||
SignalDelegation->RegisterTLSState(ParentThread);
|
||||
|
||||
@@ -384,7 +384,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
return -ENOEXEC;
|
||||
}
|
||||
|
||||
RunAsHost(SignalDelegation, Loader.DefaultRIP(), Loader.GetStackPointer(), &State);
|
||||
RunAsHost(SignalDelegation, Loader.DefaultRIP(), &State);
|
||||
SignalDelegation->UninstallTLSState(&ThreadStateObject);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -282,8 +282,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, uintptr_t InitialRip, uintptr_t StackPointer,
|
||||
FEXCore::Core::CPUState* OutputState) {
|
||||
void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, uintptr_t InitialRip, FEXCore::Core::CPUState* OutputState) {
|
||||
x86HostRunner runner;
|
||||
SignalDelegation->RegisterHostSignalHandler(
|
||||
SIGSEGV,
|
||||
@@ -295,8 +294,7 @@ void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, u
|
||||
runner.Dispatch(InitialRip);
|
||||
}
|
||||
#else
|
||||
void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, uintptr_t InitialRip, uintptr_t StackPointer,
|
||||
FEXCore::Core::CPUState* OutputState) {
|
||||
void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, uintptr_t InitialRip, FEXCore::Core::CPUState* OutputState) {
|
||||
LOGMAN_MSG_A_FMT("RunAsHost doesn't exist for this host");
|
||||
}
|
||||
#endif
|
||||
@@ -18,5 +18,4 @@ namespace FEX::HLE {
|
||||
class SignalDelegator;
|
||||
}
|
||||
|
||||
void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, uintptr_t InitialRip, uintptr_t StackPointer,
|
||||
FEXCore::Core::CPUState* OutputState);
|
||||
void RunAsHost(fextl::unique_ptr<FEX::HLE::SignalDelegator>& SignalDelegation, uintptr_t InitialRip, FEXCore::Core::CPUState* OutputState);
|
||||
@@ -23,7 +23,8 @@ target_link_libraries(arm64ecfex
|
||||
ntdll_ex
|
||||
)
|
||||
|
||||
target_link_options(arm64ecfex PRIVATE -static -nostdlib -nostartfiles -nodefaultlibs -lc++ -lc++abi -lunwind -lclang_rt.builtins-arm64ec)
|
||||
target_link_options(arm64ecfex PRIVATE -static -nostdlib -nostartfiles -nodefaultlibs -lc++ -lc++abi -lunwind)
|
||||
target_link_libraries(arm64ecfex PRIVATE ${LIBGCC_PATH})
|
||||
install(TARGETS arm64ecfex
|
||||
RUNTIME
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
|
||||
@@ -19,6 +19,10 @@ function(patch_library_wine target)
|
||||
)
|
||||
endfunction()
|
||||
|
||||
execute_process(COMMAND ${CMAKE_CXX_COMPILER} ${CMAKE_CXX_FLAGS} -print-libgcc-file-name
|
||||
OUTPUT_VARIABLE LIBGCC_PATH
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
|
||||
build_implib(ntdll)
|
||||
build_implib(wow64)
|
||||
|
||||
|
||||
@@ -351,6 +351,10 @@ DLLEXPORT_FUNC(void, GetSystemTimeAsFileTime, (LPFILETIME lpSystemTimeAsFileTime
|
||||
lpSystemTimeAsFileTime->dwHighDateTime = Time.HighPart;
|
||||
}
|
||||
|
||||
DLLEXPORT_FUNC(void, GetSystemTimePreciseAsFileTime, (LPFILETIME lpSystemTimeAsFileTime)) {
|
||||
GetSystemTimeAsFileTime(lpSystemTimeAsFileTime);
|
||||
}
|
||||
|
||||
DLLEXPORT_FUNC(WINBOOL, SetCurrentDirectoryA, (LPCSTR lpPathName)) {
|
||||
UNIMPLEMENTED();
|
||||
}
|
||||
|
||||
@@ -23,7 +23,8 @@ target_link_libraries(wow64fex
|
||||
ntdll_ex
|
||||
)
|
||||
|
||||
target_link_options(wow64fex PRIVATE -static -nostdlib -nostartfiles -nodefaultlibs -lc++ -lc++abi -lunwind -lclang_rt.builtins-aarch64)
|
||||
target_link_options(wow64fex PRIVATE -static -nostdlib -nostartfiles -nodefaultlibs -lc++ -lc++abi -lunwind)
|
||||
target_link_libraries(wow64fex PRIVATE ${LIBGCC_PATH})
|
||||
install(TARGETS wow64fex
|
||||
RUNTIME
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
|
||||
@@ -24,7 +24,10 @@ if (CMAKE_CURRENT_SOURCE_DIR STREQUAL CMAKE_SOURCE_DIR)
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
|
||||
# This gets passed in from the main cmake project
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
if (NOT DATA_DIRECTORY)
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
|
||||
endif()
|
||||
|
||||
set(TARGET_TYPE SHARED)
|
||||
set(GENERATE_GUEST_INSTALL_TARGETS TRUE)
|
||||
@@ -71,10 +74,10 @@ function(generate NAME SOURCE_FILE)
|
||||
|
||||
if (BITNESS EQUAL 32)
|
||||
set(BITNESS_FLAGS "-for-32bit-guest")
|
||||
set(BITNESS_FLAGS2 "-m32" "--target=i686-linux-unknown" "-isystem" "/usr/i686-linux-gnu/include/")
|
||||
set(BITNESS_FLAGS2 "-m32" "--target=i686-linux-gnu" "-isystem" "/usr/i686-linux-gnu/include/")
|
||||
else()
|
||||
set(BITNESS_FLAGS "")
|
||||
set(BITNESS_FLAGS2 "--target=x86_64-linux-unknown" "-isystem" "/usr/x86_64-linux-gnu/include/")
|
||||
set(BITNESS_FLAGS2 "--target=x86_64-linux-gnu" "-isystem" "/usr/x86_64-linux-gnu/include/")
|
||||
endif()
|
||||
|
||||
add_custom_command(
|
||||
|
||||
@@ -4,7 +4,10 @@ include(${FEX_PROJECT_SOURCE_DIR}/CMakeFiles/version_to_variables.cmake)
|
||||
include(GNUInstallDirs)
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set (HOSTLIBS_DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/fex-emu" CACHE PATH "global data directory")
|
||||
set (HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
if (NOT HOSTLIBS_DATA_DIRECTORY)
|
||||
set(HOSTLIBS_DATA_DIRECTORY "${CMAKE_INSTALL_FULL_LIBDIR}/fex-emu")
|
||||
endif()
|
||||
option(ENABLE_CLANG_THUNKS "Enable building thunks with clang" FALSE)
|
||||
|
||||
if (ENABLE_CLANG_THUNKS)
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# FEX-2505
|
||||
# FEX-2506
|
||||
|
||||
## FEXCore
|
||||
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "123",
|
||||
"RBX": "456",
|
||||
"RCX": "123"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug merging pops to the same register
|
||||
|
||||
; Push some stuff
|
||||
mov rsp, 0xe0000010
|
||||
mov rax, 123
|
||||
mov rbx, 456
|
||||
push rax
|
||||
push rbx
|
||||
|
||||
; Pop into the same register
|
||||
pop rcx
|
||||
pop rcx
|
||||
|
||||
; rcx now equals rax
|
||||
hlt
|
||||
@@ -0,0 +1,35 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RCX": "0x0"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug where `mov ah, 0` and `xor ah, ah` would zero the wrong register
|
||||
; subpart.
|
||||
|
||||
mov al, 127
|
||||
mov ah, 234
|
||||
mov ah, 0
|
||||
|
||||
cmp al, 127
|
||||
jne fexi_fexi_im_so_broken
|
||||
cmp ah, 0
|
||||
jne fexi_fexi_im_so_broken
|
||||
|
||||
mov al, 127
|
||||
mov ah, 234
|
||||
xor ah, ah
|
||||
|
||||
cmp al, 127
|
||||
jne fexi_fexi_im_so_broken
|
||||
cmp ah, 0
|
||||
jne fexi_fexi_im_so_broken
|
||||
|
||||
mov ecx, 0
|
||||
hlt
|
||||
|
||||
fexi_fexi_im_so_broken:
|
||||
mov ecx, 0xdeadbeef
|
||||
hlt
|
||||
@@ -59,6 +59,9 @@ TEST_CASE("futimesat - valid - null") {
|
||||
timespec time {};
|
||||
REQUIRE(clock_gettime(CLOCK_REALTIME, &time) == 0);
|
||||
|
||||
// Remove the nanoseconds to ensure consistent time setting.
|
||||
time.tv_nsec = 0;
|
||||
|
||||
// Sets the time to "Now".
|
||||
REQUIRE(compat_futimesat(fd, nullptr, nullptr) == 0);
|
||||
REQUIRE(unlinkat(AT_FDCWD, file, 0) != -1);
|
||||
|
||||
@@ -612,25 +612,23 @@
|
||||
]
|
||||
},
|
||||
"andn eax, ebx, ecx": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Comment": [
|
||||
"Map 2 0b00 0xf2 32-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"bic w26, w7, w6",
|
||||
"cmp w26, #0x0 (0)",
|
||||
"mov x4, x26"
|
||||
"bic w4, w7, w6",
|
||||
"subs w26, w4, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"andn rax, rbx, rcx": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Comment": [
|
||||
"Map 2 0b00 0xf2 64-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"bic x26, x7, x6",
|
||||
"cmp x26, #0x0 (0)",
|
||||
"mov x4, x26"
|
||||
"bic x4, x7, x6",
|
||||
"subs x26, x4, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"bzhi eax, ebx, ecx": {
|
||||
|
||||
@@ -80,8 +80,8 @@
|
||||
"Comment": "0x09",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldsetal w7, w20, [x4]",
|
||||
"orr w26, w20, w7",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w20, w7",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock adc byte [rax], cl": {
|
||||
@@ -91,7 +91,7 @@
|
||||
"cinc w20, w7, lo",
|
||||
"ldaddalb w20, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxtb w21, w7",
|
||||
"uxtb x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"add w23, w20, w22",
|
||||
"uxtb w26, w23",
|
||||
@@ -115,7 +115,7 @@
|
||||
"cinc w20, w7, lo",
|
||||
"ldaddalh w20, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxth w21, w7",
|
||||
"uxth x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"add w23, w20, w22",
|
||||
"uxth w26, w23",
|
||||
@@ -157,7 +157,7 @@
|
||||
"ldaddalb w1, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxtb w20, w20",
|
||||
"uxtb w21, w7",
|
||||
"uxtb x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"sub w23, w20, w22",
|
||||
"uxtb w26, w23",
|
||||
@@ -183,7 +183,7 @@
|
||||
"ldaddalh w1, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxth w20, w20",
|
||||
"uxth w21, w7",
|
||||
"uxth x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"sub w23, w20, w22",
|
||||
"uxth w26, w23",
|
||||
@@ -312,8 +312,8 @@
|
||||
"Comment": "0x31",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldeoral w7, w20, [x4]",
|
||||
"eor w26, w20, w7",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w20, w7",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock add qword [rax], rcx": {
|
||||
@@ -619,8 +619,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldsetal w20, w20, [x4]",
|
||||
"orr w26, w20, #0x100",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w20, #0x100",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or dword [rax], 0xFFFFFFFF": {
|
||||
@@ -629,8 +629,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffffffff",
|
||||
"ldsetal w20, w21, [x4]",
|
||||
"orr w26, w21, w20",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w21, w20",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or qword [rax], 0x100": {
|
||||
@@ -639,8 +639,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldsetal x20, x20, [x4]",
|
||||
"orr x26, x20, #0x100",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"orr x20, x20, #0x100",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or qword [rax], -2147483647": {
|
||||
@@ -649,8 +649,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, #0xffffffff80000001",
|
||||
"ldsetal x20, x20, [x4]",
|
||||
"orr x26, x20, #0xffffffff80000001",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"orr x20, x20, #0xffffffff80000001",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or word [rax], 1": {
|
||||
@@ -672,8 +672,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldsetal w20, w20, [x4]",
|
||||
"orr w26, w20, #0x1",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w20, #0x1",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or qword [rax], 1": {
|
||||
@@ -682,12 +682,12 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldsetal x20, x20, [x4]",
|
||||
"orr x26, x20, #0x1",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"orr x20, x20, #0x1",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock adc byte [rax], 1": {
|
||||
"ExpectedInstructionCount": 17,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Comment": "GROUP1 0x80 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -701,16 +701,14 @@
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
"eor w20, w27, #0x1",
|
||||
"eor w22, w26, w27",
|
||||
"bic w20, w22, w20",
|
||||
"bic w20, w26, w27",
|
||||
"ubfx x20, x20, #7, #1",
|
||||
"bfi w21, w20, #28, #1",
|
||||
"msr nzcv, x21"
|
||||
]
|
||||
},
|
||||
"lock adc byte [rax], 0xFF": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"ExpectedInstructionCount": 16,
|
||||
"Comment": "GROUP1 0x80 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xff",
|
||||
@@ -725,16 +723,14 @@
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0xff",
|
||||
"eor w21, w26, w21",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"ubfx x20, x20, #7, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
]
|
||||
},
|
||||
"lock adc word [rax], 0x100": {
|
||||
"ExpectedInstructionCount": 17,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Comment": "GROUP1 0x81 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
@@ -748,16 +744,14 @@
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
"eor w20, w27, #0x100",
|
||||
"eor w22, w26, w27",
|
||||
"bic w20, w22, w20",
|
||||
"bic w20, w26, w27",
|
||||
"ubfx x20, x20, #15, #1",
|
||||
"bfi w21, w20, #28, #1",
|
||||
"msr nzcv, x21"
|
||||
]
|
||||
},
|
||||
"lock adc word [rax], 0xFFFF": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"ExpectedInstructionCount": 16,
|
||||
"Comment": "GROUP1 0x81 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffff",
|
||||
@@ -772,9 +766,7 @@
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0xffff",
|
||||
"eor w21, w26, w21",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"ubfx x20, x20, #15, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
@@ -846,7 +838,7 @@
|
||||
]
|
||||
},
|
||||
"lock adc word [rax], 1": {
|
||||
"ExpectedInstructionCount": 17,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Comment": "GROUP1 0x83 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -860,9 +852,7 @@
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
"eor w20, w27, #0x1",
|
||||
"eor w22, w26, w27",
|
||||
"bic w20, w22, w20",
|
||||
"bic w20, w26, w27",
|
||||
"ubfx x20, x20, #15, #1",
|
||||
"bfi w21, w20, #28, #1",
|
||||
"msr nzcv, x21"
|
||||
@@ -901,7 +891,7 @@
|
||||
]
|
||||
},
|
||||
"lock sbb byte [rax], 1": {
|
||||
"ExpectedInstructionCount": 19,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Comment": "GROUP1 0x80 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -917,16 +907,14 @@
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0x1",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"ubfx x20, x20, #7, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
]
|
||||
},
|
||||
"lock sbb byte [rax], 0xFF": {
|
||||
"ExpectedInstructionCount": 20,
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": "GROUP1 0x80 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xff",
|
||||
@@ -943,16 +931,14 @@
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0xff",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w26, w21",
|
||||
"ubfx x20, x20, #7, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
]
|
||||
},
|
||||
"lock sbb word [rax], 0x100": {
|
||||
"ExpectedInstructionCount": 19,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Comment": "GROUP1 0x81 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
@@ -968,16 +954,14 @@
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0x100",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"ubfx x20, x20, #15, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
]
|
||||
},
|
||||
"lock sbb word [rax], 0xFFFF": {
|
||||
"ExpectedInstructionCount": 20,
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": "GROUP1 0x81 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffff",
|
||||
@@ -994,9 +978,7 @@
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0xffff",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w26, w21",
|
||||
"ubfx x20, x20, #15, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
@@ -1048,7 +1030,7 @@
|
||||
]
|
||||
},
|
||||
"lock sbb word [rax], 1": {
|
||||
"ExpectedInstructionCount": 19,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Comment": "GROUP1 0x83 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -1064,9 +1046,7 @@
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"mrs x22, nzcv",
|
||||
"bfi w22, w20, #29, #1",
|
||||
"eor w20, w21, #0x1",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"ubfx x20, x20, #15, #1",
|
||||
"bfi w22, w20, #28, #1",
|
||||
"msr nzcv, x22"
|
||||
@@ -1423,8 +1403,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldeoral w20, w20, [x4]",
|
||||
"eor w26, w20, #0x100",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w20, #0x100",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor dword [rax], 0xFFFFFFFF": {
|
||||
@@ -1433,8 +1413,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffffffff",
|
||||
"ldeoral w20, w21, [x4]",
|
||||
"eor w26, w21, w20",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w21, w20",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor qword [rax], 0x100": {
|
||||
@@ -1443,8 +1423,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldeoral x20, x20, [x4]",
|
||||
"eor x26, x20, #0x100",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"eor x20, x20, #0x100",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor qword [rax], -2147483647": {
|
||||
@@ -1453,8 +1433,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, #0xffffffff80000001",
|
||||
"ldeoral x20, x20, [x4]",
|
||||
"eor x26, x20, #0xffffffff80000001",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"eor x20, x20, #0xffffffff80000001",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor word [rax], 1": {
|
||||
@@ -1476,8 +1456,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldeoral w20, w20, [x4]",
|
||||
"eor w26, w20, #0x1",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w20, #0x1",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor qword [rax], 1": {
|
||||
@@ -1486,8 +1466,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldeoral x20, x20, [x4]",
|
||||
"eor x26, x20, #0x1",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"eor x20, x20, #0x1",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock dec byte [rax]": {
|
||||
|
||||
@@ -17,6 +17,343 @@
|
||||
"These are instruction combinations that could be more optimal if FEX optimized for them"
|
||||
],
|
||||
"Instructions": {
|
||||
"cpuid constant": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 57,
|
||||
"Comment": [
|
||||
"CPUID function call with constant function id"
|
||||
],
|
||||
"x86Insts": [
|
||||
"mov rax, 0",
|
||||
"cpuid"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w4, #0x0",
|
||||
"isb",
|
||||
"mov x1, x4",
|
||||
"mov x2, x7",
|
||||
"sub sp, sp, #0xf0 (240)",
|
||||
"mov x3, sp",
|
||||
"st1 {v2.2d, v3.2d}, [x3], #32",
|
||||
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x3], #64",
|
||||
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x3], #64",
|
||||
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x3], #64",
|
||||
"stp x18, x30, [x3], #16",
|
||||
"mrs x3, nzcv",
|
||||
"str w3, [x28, #1000]",
|
||||
"stp x4, x7, [x28, #288]",
|
||||
"stp x5, x6, [x28, #304]",
|
||||
"stp x8, x9, [x28, #320]",
|
||||
"stp x10, x11, [x28, #336]",
|
||||
"stp x12, x13, [x28, #352]",
|
||||
"stp x14, x15, [x28, #368]",
|
||||
"stp x16, x17, [x28, #384]",
|
||||
"stp x19, x29, [x28, #400]",
|
||||
"stp w26, w27, [x28, #16]",
|
||||
"add x3, x28, #0x1a0 (416)",
|
||||
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x3], #64",
|
||||
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x3], #64",
|
||||
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x3], #64",
|
||||
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x3], #64",
|
||||
"ldr x0, [x28, #1368]",
|
||||
"ldr x3, [x28, #1376]",
|
||||
"blr x3",
|
||||
"ldr w4, [x28, #1000]",
|
||||
"msr nzcv, x4",
|
||||
"add x4, x28, #0x1a0 (416)",
|
||||
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x4], #64",
|
||||
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x4], #64",
|
||||
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x4], #64",
|
||||
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x4], #64",
|
||||
"ldp x4, x7, [x28, #288]",
|
||||
"ldp x5, x6, [x28, #304]",
|
||||
"ldp x8, x9, [x28, #320]",
|
||||
"ldp x10, x11, [x28, #336]",
|
||||
"ldp x12, x13, [x28, #352]",
|
||||
"ldp x14, x15, [x28, #368]",
|
||||
"ldp x16, x17, [x28, #384]",
|
||||
"ldp x19, x29, [x28, #400]",
|
||||
"ldp w26, w27, [x28, #16]",
|
||||
"ld1 {v2.2d, v3.2d}, [sp], #32",
|
||||
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
|
||||
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
|
||||
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
|
||||
"ldp x18, x30, [sp], #16",
|
||||
"mov w20, w0",
|
||||
"mov w21, w1",
|
||||
"lsr x6, x0, #32",
|
||||
"lsr x5, x1, #32",
|
||||
"mov x7, x21",
|
||||
"mov x4, x20"
|
||||
]
|
||||
},
|
||||
"xgetbv constant": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 51,
|
||||
"Comment": [
|
||||
"XGETBV function call with constant function id"
|
||||
],
|
||||
"x86Insts": [
|
||||
"mov rcx, 0",
|
||||
"xgetbv"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w7, #0x0",
|
||||
"sub sp, sp, #0xf0 (240)",
|
||||
"mov x3, sp",
|
||||
"st1 {v2.2d, v3.2d}, [x3], #32",
|
||||
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x3], #64",
|
||||
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x3], #64",
|
||||
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x3], #64",
|
||||
"stp x18, x30, [x3], #16",
|
||||
"mrs x3, nzcv",
|
||||
"str w3, [x28, #1000]",
|
||||
"stp x4, x7, [x28, #288]",
|
||||
"stp x5, x6, [x28, #304]",
|
||||
"stp x8, x9, [x28, #320]",
|
||||
"stp x10, x11, [x28, #336]",
|
||||
"stp x12, x13, [x28, #352]",
|
||||
"stp x14, x15, [x28, #368]",
|
||||
"stp x16, x17, [x28, #384]",
|
||||
"stp x19, x29, [x28, #400]",
|
||||
"stp w26, w27, [x28, #16]",
|
||||
"add x3, x28, #0x1a0 (416)",
|
||||
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x3], #64",
|
||||
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x3], #64",
|
||||
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x3], #64",
|
||||
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x3], #64",
|
||||
"mov w1, w7",
|
||||
"ldr x0, [x28, #1368]",
|
||||
"ldr x2, [x28, #1384]",
|
||||
"blr x2",
|
||||
"ldr w4, [x28, #1000]",
|
||||
"msr nzcv, x4",
|
||||
"add x4, x28, #0x1a0 (416)",
|
||||
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x4], #64",
|
||||
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x4], #64",
|
||||
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x4], #64",
|
||||
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x4], #64",
|
||||
"ldp x4, x7, [x28, #288]",
|
||||
"ldp x5, x6, [x28, #304]",
|
||||
"ldp x8, x9, [x28, #320]",
|
||||
"ldp x10, x11, [x28, #336]",
|
||||
"ldp x12, x13, [x28, #352]",
|
||||
"ldp x14, x15, [x28, #368]",
|
||||
"ldp x16, x17, [x28, #384]",
|
||||
"ldp x19, x29, [x28, #400]",
|
||||
"ldp w26, w27, [x28, #16]",
|
||||
"ld1 {v2.2d, v3.2d}, [sp], #32",
|
||||
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
|
||||
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
|
||||
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
|
||||
"ldp x18, x30, [sp], #16",
|
||||
"mov w4, w0",
|
||||
"lsr x5, x0, #32"
|
||||
]
|
||||
},
|
||||
"signed div narrow": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 29,
|
||||
"Comment": [
|
||||
"div narrowing with known smaller sources",
|
||||
"dividend in rdx:rax"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cqo",
|
||||
"idiv rcx"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"asr x5, x4, #63",
|
||||
"asr x0, x4, #63",
|
||||
"eor x0, x0, x5",
|
||||
"cbz x0, #+0x28",
|
||||
"mov x0, x5",
|
||||
"mov x1, x4",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #3400]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
"mov x20, x0",
|
||||
"b #+0x8",
|
||||
"sdiv x20, x4, x7",
|
||||
"asr x0, x4, #63",
|
||||
"eor x0, x0, x5",
|
||||
"cbz x0, #+0x28",
|
||||
"mov x0, x5",
|
||||
"mov x1, x4",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #3416]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
"mov x5, x0",
|
||||
"b #+0xc",
|
||||
"sdiv x0, x4, x7",
|
||||
"msub x5, x0, x7, x4",
|
||||
"mov x4, x20"
|
||||
]
|
||||
},
|
||||
"unsigned div narrow": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 25,
|
||||
"Comment": [
|
||||
"div narrowing with known smaller sources",
|
||||
"dividend in rdx:rax"
|
||||
],
|
||||
"x86Insts": [
|
||||
"mov rdx, 0",
|
||||
"div rcx"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w5, #0x0",
|
||||
"cbz x5, #+0x28",
|
||||
"mov x0, x5",
|
||||
"mov x1, x4",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #3392]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
"mov x20, x0",
|
||||
"b #+0x8",
|
||||
"udiv x20, x4, x7",
|
||||
"cbz x5, #+0x28",
|
||||
"mov x0, x5",
|
||||
"mov x1, x4",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #3408]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
"mov x5, x0",
|
||||
"b #+0xc",
|
||||
"udiv x0, x4, x7",
|
||||
"msub x5, x0, x7, x4",
|
||||
"mov x4, x20"
|
||||
]
|
||||
},
|
||||
"inline syscall": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 106,
|
||||
"Comment": [
|
||||
"Simple inline syscall check",
|
||||
"When the syscall number is a known constant that matches host semantics then it can be inlined",
|
||||
"InstcountCI is setup that it claims to be a 64-bit Linux syscall handler"
|
||||
],
|
||||
"x86Insts": [
|
||||
"mov rax, 0",
|
||||
"syscall"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w4, #0x0",
|
||||
"mov w20, #0x5",
|
||||
"movk w20, #0x1, lsl #16",
|
||||
"str x20, [x28, #24]",
|
||||
"cset w20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
"ldrb w21, [x28, #984]",
|
||||
"orr x20, x20, x21, lsl #8",
|
||||
"ldrb w21, [x28, #985]",
|
||||
"orr x20, x20, x21, lsl #9",
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
"ldrb w21, [x28, #990]",
|
||||
"orr x20, x20, x21, lsl #14",
|
||||
"ldrb w21, [x28, #992]",
|
||||
"orr x20, x20, x21, lsl #16",
|
||||
"ldrb w21, [x28, #993]",
|
||||
"orr x20, x20, x21, lsl #17",
|
||||
"ldrb w21, [x28, #994]",
|
||||
"orr x20, x20, x21, lsl #18",
|
||||
"ldrb w21, [x28, #995]",
|
||||
"orr x20, x20, x21, lsl #19",
|
||||
"ldrb w21, [x28, #996]",
|
||||
"orr x20, x20, x21, lsl #20",
|
||||
"ldrb w21, [x28, #997]",
|
||||
"orr x20, x20, x21, lsl #21",
|
||||
"eor w0, w26, w26, lsr #4",
|
||||
"eor w0, w0, w0, lsr #2",
|
||||
"eor w21, w0, w0, lsr #1",
|
||||
"orr x21, x21, #0xfffffffffffffffe",
|
||||
"orn x20, x20, x21, ror #62",
|
||||
"mrs x21, nzcv",
|
||||
"and x21, x21, #0xc0000000",
|
||||
"orr x20, x20, x21, lsr #24",
|
||||
"orr x15, x20, #0x2",
|
||||
"mov w7, #0x7",
|
||||
"movk w7, #0x1, lsl #16",
|
||||
"sub sp, sp, #0xf0 (240)",
|
||||
"mov x0, sp",
|
||||
"st1 {v2.2d, v3.2d}, [x0], #32",
|
||||
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
|
||||
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
|
||||
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
|
||||
"stp x18, x30, [x0], #16",
|
||||
"mrs x0, nzcv",
|
||||
"str w0, [x28, #1000]",
|
||||
"stp x4, x7, [x28, #288]",
|
||||
"stp x5, x6, [x28, #304]",
|
||||
"stp x8, x9, [x28, #320]",
|
||||
"stp x10, x11, [x28, #336]",
|
||||
"stp x12, x13, [x28, #352]",
|
||||
"stp x14, x15, [x28, #368]",
|
||||
"stp x16, x17, [x28, #384]",
|
||||
"stp x19, x29, [x28, #400]",
|
||||
"stp w26, w27, [x28, #16]",
|
||||
"add x0, x28, #0x1a0 (416)",
|
||||
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
|
||||
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
|
||||
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
|
||||
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
|
||||
"mov w0, #0xffff",
|
||||
"str x0, [x28, #1312]",
|
||||
"sub sp, sp, #0x40 (64)",
|
||||
"str x4, [sp]",
|
||||
"str x11, [sp, #8]",
|
||||
"str x10, [sp, #16]",
|
||||
"str x5, [sp, #24]",
|
||||
"str x14, [sp, #32]",
|
||||
"str x12, [sp, #40]",
|
||||
"str x13, [sp, #48]",
|
||||
"ldr x0, [x28, #1392]",
|
||||
"ldr x3, [x28, #1400]",
|
||||
"mov x1, x28",
|
||||
"mov x2, sp",
|
||||
"blr x3",
|
||||
"add sp, sp, #0x40 (64)",
|
||||
"ldr w1, [x28, #1000]",
|
||||
"msr nzcv, x1",
|
||||
"add x1, x28, #0x1a0 (416)",
|
||||
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x1], #64",
|
||||
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x1], #64",
|
||||
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x1], #64",
|
||||
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x1], #64",
|
||||
"ldp x4, x7, [x28, #288]",
|
||||
"ldp x5, x6, [x28, #304]",
|
||||
"ldp x8, x9, [x28, #320]",
|
||||
"ldp x10, x11, [x28, #336]",
|
||||
"ldp x12, x13, [x28, #352]",
|
||||
"ldp x14, x15, [x28, #368]",
|
||||
"ldp x16, x17, [x28, #384]",
|
||||
"ldp x19, x29, [x28, #400]",
|
||||
"ldp w26, w27, [x28, #16]",
|
||||
"str xzr, [x28, #1312]",
|
||||
"ld1 {v2.2d, v3.2d}, [sp], #32",
|
||||
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
|
||||
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
|
||||
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
|
||||
"ldp x18, x30, [sp], #16",
|
||||
"mov x4, x0"
|
||||
]
|
||||
},
|
||||
"push ax, bx": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 2,
|
||||
@@ -34,7 +371,7 @@
|
||||
},
|
||||
"push rax, rbx": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 2,
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"Mergable 64-bit pushes"
|
||||
],
|
||||
@@ -43,8 +380,7 @@
|
||||
"push rbx"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"str x4, [x8, #-8]!",
|
||||
"str x6, [x8, #-8]!"
|
||||
"stp x6, x4, [x8, #-16]!"
|
||||
]
|
||||
},
|
||||
"adds xmm0, xmm1, xmm2": {
|
||||
@@ -226,7 +562,7 @@
|
||||
},
|
||||
"positive rep movsb": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 44,
|
||||
"ExpectedInstructionCount": 43,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -236,8 +572,7 @@
|
||||
"rep movsb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"mov w20, #0x1",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
@@ -274,17 +609,17 @@
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x7",
|
||||
"add x23, x0, x2",
|
||||
"add x22, x1, x2",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x11, x23",
|
||||
"mov x10, x22",
|
||||
"mov x7, x20"
|
||||
"add x22, x0, x2",
|
||||
"add x21, x1, x2",
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]",
|
||||
"mov x11, x22",
|
||||
"mov x10, x21"
|
||||
]
|
||||
},
|
||||
"positive rep movsw": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 44,
|
||||
"ExpectedInstructionCount": 43,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -294,8 +629,7 @@
|
||||
"rep movsw"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"mov w20, #0x1",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
@@ -332,17 +666,17 @@
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x7",
|
||||
"add x23, x0, x2, lsl #1",
|
||||
"add x22, x1, x2, lsl #1",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x11, x23",
|
||||
"mov x10, x22",
|
||||
"mov x7, x20"
|
||||
"add x22, x0, x2, lsl #1",
|
||||
"add x21, x1, x2, lsl #1",
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]",
|
||||
"mov x11, x22",
|
||||
"mov x10, x21"
|
||||
]
|
||||
},
|
||||
"positive rep movsd": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 44,
|
||||
"ExpectedInstructionCount": 43,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -352,8 +686,7 @@
|
||||
"rep movsd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"mov w20, #0x1",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
@@ -390,17 +723,17 @@
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x7",
|
||||
"add x23, x0, x2, lsl #2",
|
||||
"add x22, x1, x2, lsl #2",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x11, x23",
|
||||
"mov x10, x22",
|
||||
"mov x7, x20"
|
||||
"add x22, x0, x2, lsl #2",
|
||||
"add x21, x1, x2, lsl #2",
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]",
|
||||
"mov x11, x22",
|
||||
"mov x10, x21"
|
||||
]
|
||||
},
|
||||
"positive rep movsq": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 44,
|
||||
"ExpectedInstructionCount": 43,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -410,8 +743,7 @@
|
||||
"rep movsq"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"mov w20, #0x1",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
@@ -448,12 +780,12 @@
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x7",
|
||||
"add x23, x0, x2, lsl #3",
|
||||
"add x22, x1, x2, lsl #3",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x11, x23",
|
||||
"mov x10, x22",
|
||||
"mov x7, x20"
|
||||
"add x22, x0, x2, lsl #3",
|
||||
"add x21, x1, x2, lsl #3",
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]",
|
||||
"mov x11, x22",
|
||||
"mov x10, x21"
|
||||
]
|
||||
},
|
||||
"negative rep movsb": {
|
||||
@@ -702,7 +1034,7 @@
|
||||
},
|
||||
"positive rep stosb": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 30,
|
||||
"ExpectedInstructionCount": 29,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -712,15 +1044,14 @@
|
||||
"rep stosb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"uxtb w22, w4",
|
||||
"mov w20, #0x1",
|
||||
"uxtb w21, w4",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x58",
|
||||
"sub x0, x0, #0x20 (32)",
|
||||
"tbnz x0, #63, #+0x3c",
|
||||
"dup v1.16b, w22",
|
||||
"dup v1.16b, w21",
|
||||
"sub x0, x0, #0x20 (32)",
|
||||
"tbnz x0, #63, #+0x14",
|
||||
"stp q1, q1, [x1], #32",
|
||||
@@ -736,17 +1067,17 @@
|
||||
"tbz x0, #63, #-0x8",
|
||||
"add x0, x0, #0x20 (32)",
|
||||
"cbz x0, #+0x10",
|
||||
"strb w22, [x1], #1",
|
||||
"strb w21, [x1], #1",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x7",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x7, x20"
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]"
|
||||
]
|
||||
},
|
||||
"positive rep stosw": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 30,
|
||||
"ExpectedInstructionCount": 29,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -756,15 +1087,14 @@
|
||||
"rep stosw"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"uxth w22, w4",
|
||||
"mov w20, #0x1",
|
||||
"uxth w21, w4",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x58",
|
||||
"sub x0, x0, #0x10 (16)",
|
||||
"tbnz x0, #63, #+0x3c",
|
||||
"dup v1.8h, w22",
|
||||
"dup v1.8h, w21",
|
||||
"sub x0, x0, #0x10 (16)",
|
||||
"tbnz x0, #63, #+0x14",
|
||||
"stp q1, q1, [x1], #32",
|
||||
@@ -780,17 +1110,17 @@
|
||||
"tbz x0, #63, #-0x8",
|
||||
"add x0, x0, #0x10 (16)",
|
||||
"cbz x0, #+0x10",
|
||||
"strh w22, [x1], #2",
|
||||
"strh w21, [x1], #2",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x7, lsl #1",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x7, x20"
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]"
|
||||
]
|
||||
},
|
||||
"positive rep stosd": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 30,
|
||||
"ExpectedInstructionCount": 29,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -800,15 +1130,14 @@
|
||||
"rep stosd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"mov w22, w4",
|
||||
"mov w20, #0x1",
|
||||
"mov w21, w4",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x58",
|
||||
"sub x0, x0, #0x8 (8)",
|
||||
"tbnz x0, #63, #+0x3c",
|
||||
"dup v1.4s, w22",
|
||||
"dup v1.4s, w21",
|
||||
"sub x0, x0, #0x8 (8)",
|
||||
"tbnz x0, #63, #+0x14",
|
||||
"stp q1, q1, [x1], #32",
|
||||
@@ -824,17 +1153,17 @@
|
||||
"tbz x0, #63, #-0x8",
|
||||
"add x0, x0, #0x8 (8)",
|
||||
"cbz x0, #+0x10",
|
||||
"str w22, [x1], #4",
|
||||
"str w21, [x1], #4",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x7, lsl #2",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x7, x20"
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]"
|
||||
]
|
||||
},
|
||||
"positive rep stosq": {
|
||||
"x86InstructionCount": 2,
|
||||
"ExpectedInstructionCount": 29,
|
||||
"ExpectedInstructionCount": 28,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -844,8 +1173,7 @@
|
||||
"rep stosq"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"mov w21, #0x1",
|
||||
"mov w20, #0x1",
|
||||
"mov x0, x7",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x58",
|
||||
@@ -871,8 +1199,8 @@
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x7, lsl #3",
|
||||
"strb w21, [x28, #986]",
|
||||
"mov x7, x20"
|
||||
"mov w7, #0x0",
|
||||
"strb w20, [x28, #986]"
|
||||
]
|
||||
},
|
||||
"negative rep stosb": {
|
||||
@@ -1056,7 +1384,7 @@
|
||||
},
|
||||
"Sekiro spill block": {
|
||||
"x86InstructionCount": 119,
|
||||
"ExpectedInstructionCount": 126,
|
||||
"ExpectedInstructionCount": 118,
|
||||
"Comment": [
|
||||
"This block of code came from the settings screen when it loaded",
|
||||
"It was originally at RIP: 0x14232cca0 and has been deobfuscated"
|
||||
@@ -1184,14 +1512,10 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"str x7, [x8, #8]",
|
||||
"str x6, [x8, #-8]!",
|
||||
"str x9, [x8, #-8]!",
|
||||
"str x10, [x8, #-8]!",
|
||||
"str x11, [x8, #-8]!",
|
||||
"str x16, [x8, #-8]!",
|
||||
"str x17, [x8, #-8]!",
|
||||
"str x19, [x8, #-8]!",
|
||||
"str x29, [x8, #-8]!",
|
||||
"stp x9, x6, [x8, #-16]!",
|
||||
"stp x11, x10, [x8, #-16]!",
|
||||
"stp x17, x16, [x8, #-16]!",
|
||||
"stp x29, x19, [x8, #-16]!",
|
||||
"sub x8, x8, #0x18 (24)",
|
||||
"ldr w7, [x5, #36]",
|
||||
"ldr w10, [x5]",
|
||||
@@ -1300,14 +1624,10 @@
|
||||
"mvn w27, w8",
|
||||
"adds x26, x8, #0x18 (24)",
|
||||
"mov x8, x26",
|
||||
"ldr x29, [x8], #8",
|
||||
"ldr x19, [x8], #8",
|
||||
"ldr x17, [x8], #8",
|
||||
"ldr x16, [x8], #8",
|
||||
"ldr x11, [x8], #8",
|
||||
"ldr x10, [x8], #8",
|
||||
"ldr x9, [x8], #8",
|
||||
"ldr x6, [x8], #8",
|
||||
"ldp x29, x19, [x8], #16",
|
||||
"ldp x17, x16, [x8], #16",
|
||||
"ldp x11, x10, [x8], #16",
|
||||
"ldp x9, x6, [x8], #16",
|
||||
"cfinv"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -74,9 +74,9 @@
|
||||
"nop",
|
||||
"ldapurh w20, [x7, #24]",
|
||||
"nop",
|
||||
"bfxil w4, w20, #0, #16",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"ldapurb w20, [x7, #26]",
|
||||
"bfxil w6, w20, #0, #8"
|
||||
"bfxil x6, x20, #0, #8"
|
||||
]
|
||||
},
|
||||
"Store variables to memory": {
|
||||
|
||||
@@ -71,8 +71,8 @@
|
||||
"Comment": "0x09",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldsetal w7, w20, [x4]",
|
||||
"orr w26, w20, w7",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w20, w7",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock adc byte [rax], cl": {
|
||||
@@ -82,7 +82,7 @@
|
||||
"cinc w20, w7, lo",
|
||||
"ldaddalb w20, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxtb w21, w7",
|
||||
"uxtb x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"add w23, w20, w22",
|
||||
"uxtb w26, w23",
|
||||
@@ -103,7 +103,7 @@
|
||||
"cinc w20, w7, lo",
|
||||
"ldaddalh w20, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxth w21, w7",
|
||||
"uxth x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"add w23, w20, w22",
|
||||
"uxth w26, w23",
|
||||
@@ -138,7 +138,7 @@
|
||||
"ldaddalb w1, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxtb w20, w20",
|
||||
"uxtb w21, w7",
|
||||
"uxtb x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"sub w23, w20, w22",
|
||||
"uxtb w26, w23",
|
||||
@@ -161,7 +161,7 @@
|
||||
"ldaddalh w1, w20, [x4]",
|
||||
"eor x27, x20, x7",
|
||||
"uxth w20, w20",
|
||||
"uxth w21, w7",
|
||||
"uxth x21, w7",
|
||||
"cinc w22, w21, lo",
|
||||
"sub w23, w20, w22",
|
||||
"uxth w26, w23",
|
||||
@@ -277,8 +277,8 @@
|
||||
"Comment": "0x31",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldeoral w7, w20, [x4]",
|
||||
"eor w26, w20, w7",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w20, w7",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock add qword [rax], rcx": {
|
||||
@@ -514,8 +514,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldsetal w20, w20, [x4]",
|
||||
"orr w26, w20, #0x100",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w20, #0x100",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or dword [rax], 0xFFFFFFFF": {
|
||||
@@ -524,8 +524,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffffffff",
|
||||
"ldsetal w20, w21, [x4]",
|
||||
"orr w26, w21, w20",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w21, w20",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or qword [rax], 0x100": {
|
||||
@@ -534,8 +534,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldsetal x20, x20, [x4]",
|
||||
"orr x26, x20, #0x100",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"orr x20, x20, #0x100",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or qword [rax], -2147483647": {
|
||||
@@ -544,8 +544,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, #0xffffffff80000001",
|
||||
"ldsetal x20, x20, [x4]",
|
||||
"orr x26, x20, #0xffffffff80000001",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"orr x20, x20, #0xffffffff80000001",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or word [rax], 1": {
|
||||
@@ -565,8 +565,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldsetal w20, w20, [x4]",
|
||||
"orr w26, w20, #0x1",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"orr w20, w20, #0x1",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock or qword [rax], 1": {
|
||||
@@ -575,12 +575,12 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldsetal x20, x20, [x4]",
|
||||
"orr x26, x20, #0x1",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"orr x20, x20, #0x1",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock adc byte [rax], 1": {
|
||||
"ExpectedInstructionCount": 14,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": "GROUP1 0x80 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -593,14 +593,12 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w27, #0x1",
|
||||
"eor w21, w26, w27",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w26, w27",
|
||||
"rmif x20, #7, #nzcV"
|
||||
]
|
||||
},
|
||||
"lock adc byte [rax], 0xFF": {
|
||||
"ExpectedInstructionCount": 15,
|
||||
"ExpectedInstructionCount": 13,
|
||||
"Comment": "GROUP1 0x80 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xff",
|
||||
@@ -614,14 +612,12 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0xff",
|
||||
"eor w21, w26, w21",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"rmif x20, #7, #nzcV"
|
||||
]
|
||||
},
|
||||
"lock adc word [rax], 0x100": {
|
||||
"ExpectedInstructionCount": 14,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": "GROUP1 0x81 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
@@ -634,14 +630,12 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w27, #0x100",
|
||||
"eor w21, w26, w27",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w26, w27",
|
||||
"rmif x20, #15, #nzcV"
|
||||
]
|
||||
},
|
||||
"lock adc word [rax], 0xFFFF": {
|
||||
"ExpectedInstructionCount": 15,
|
||||
"ExpectedInstructionCount": 13,
|
||||
"Comment": "GROUP1 0x81 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffff",
|
||||
@@ -655,9 +649,7 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0xffff",
|
||||
"eor w21, w26, w21",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"rmif x20, #15, #nzcV"
|
||||
]
|
||||
},
|
||||
@@ -711,7 +703,7 @@
|
||||
]
|
||||
},
|
||||
"lock adc word [rax], 1": {
|
||||
"ExpectedInstructionCount": 14,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": "GROUP1 0x83 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -724,9 +716,7 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w27, #0x1",
|
||||
"eor w21, w26, w27",
|
||||
"bic w20, w21, w20",
|
||||
"bic w20, w26, w27",
|
||||
"rmif x20, #15, #nzcV"
|
||||
]
|
||||
},
|
||||
@@ -755,7 +745,7 @@
|
||||
]
|
||||
},
|
||||
"lock sbb byte [rax], 1": {
|
||||
"ExpectedInstructionCount": 16,
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": "GROUP1 0x80 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -770,14 +760,12 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0x1",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"rmif x20, #7, #nzcV"
|
||||
]
|
||||
},
|
||||
"lock sbb byte [rax], 0xFF": {
|
||||
"ExpectedInstructionCount": 17,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Comment": "GROUP1 0x80 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xff",
|
||||
@@ -793,14 +781,12 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #24",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0xff",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w26, w21",
|
||||
"rmif x20, #7, #nzcV"
|
||||
]
|
||||
},
|
||||
"lock sbb word [rax], 0x100": {
|
||||
"ExpectedInstructionCount": 16,
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": "GROUP1 0x81 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
@@ -815,14 +801,12 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0x100",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"rmif x20, #15, #nzcV"
|
||||
]
|
||||
},
|
||||
"lock sbb word [rax], 0xFFFF": {
|
||||
"ExpectedInstructionCount": 17,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Comment": "GROUP1 0x81 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffff",
|
||||
@@ -838,9 +822,7 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0xffff",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w26, w21",
|
||||
"rmif x20, #15, #nzcV"
|
||||
]
|
||||
},
|
||||
@@ -890,7 +872,7 @@
|
||||
]
|
||||
},
|
||||
"lock sbb word [rax], 1": {
|
||||
"ExpectedInstructionCount": 16,
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": "GROUP1 0x83 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
@@ -905,9 +887,7 @@
|
||||
"cset x20, hs",
|
||||
"cmn wzr, w26, lsl #16",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"eor w20, w21, #0x1",
|
||||
"eor w21, w26, w21",
|
||||
"and w20, w21, w20",
|
||||
"bic w20, w21, w26",
|
||||
"rmif x20, #15, #nzcV"
|
||||
]
|
||||
},
|
||||
@@ -1232,8 +1212,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldeoral w20, w20, [x4]",
|
||||
"eor w26, w20, #0x100",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w20, #0x100",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor dword [rax], 0xFFFFFFFF": {
|
||||
@@ -1242,8 +1222,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffffffff",
|
||||
"ldeoral w20, w21, [x4]",
|
||||
"eor w26, w21, w20",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w21, w20",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor qword [rax], 0x100": {
|
||||
@@ -1252,8 +1232,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x100",
|
||||
"ldeoral x20, x20, [x4]",
|
||||
"eor x26, x20, #0x100",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"eor x20, x20, #0x100",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor qword [rax], -2147483647": {
|
||||
@@ -1262,8 +1242,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, #0xffffffff80000001",
|
||||
"ldeoral x20, x20, [x4]",
|
||||
"eor x26, x20, #0xffffffff80000001",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"eor x20, x20, #0xffffffff80000001",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor word [rax], 1": {
|
||||
@@ -1283,8 +1263,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldeoral w20, w20, [x4]",
|
||||
"eor w26, w20, #0x1",
|
||||
"cmp w26, #0x0 (0)"
|
||||
"eor w20, w20, #0x1",
|
||||
"subs w26, w20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock xor qword [rax], 1": {
|
||||
@@ -1293,8 +1273,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldeoral x20, x20, [x4]",
|
||||
"eor x26, x20, #0x1",
|
||||
"cmp x26, #0x0 (0)"
|
||||
"eor x20, x20, #0x1",
|
||||
"subs x26, x20, #0x0 (0)"
|
||||
]
|
||||
},
|
||||
"lock dec byte [rax]": {
|
||||
|
||||
Loaded 100 of 124 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user