mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 21:00:17 +02:00
Compare commits
205
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
faed139c34 | ||
|
|
80927bf0e1 | ||
|
|
64276dbd0c | ||
|
|
b619f381b3 | ||
|
|
3b0aff5fb9 | ||
|
|
a8ab8bbe8e | ||
|
|
3e2ba6d835 | ||
|
|
c6497fe32b | ||
|
|
c8ef77c15f | ||
|
|
b35fadf7e3 | ||
|
|
250ffb6d23 | ||
|
|
f6b1434d63 | ||
|
|
6bbae69c75 | ||
|
|
d0f54bcb23 | ||
|
|
470615b896 | ||
|
|
b02ab8ee19 | ||
|
|
5f6046be4c | ||
|
|
e923e83efb | ||
|
|
e836e4212d | ||
|
|
9417c93110 | ||
|
|
068599b1ec | ||
|
|
7216415bfc | ||
|
|
3020626506 | ||
|
|
0a79fa8d5d | ||
|
|
f8380b9adb | ||
|
|
d898028bc3 | ||
|
|
f090700184 | ||
|
|
01d29dffb9 | ||
|
|
14ba64a22d | ||
|
|
6716077cb6 | ||
|
|
8892580c41 | ||
|
|
1b41304fc1 | ||
|
|
7de66ac3a4 | ||
|
|
85a1c1ff25 | ||
|
|
8e892ece59 | ||
|
|
aa1344aadd | ||
|
|
3f02d7c665 | ||
|
|
f328fca880 | ||
|
|
47d79978ef | ||
|
|
2e24f34a3f | ||
|
|
6e8af295c5 | ||
|
|
bba156a3c1 | ||
|
|
8015ce2099 | ||
|
|
1153c1a538 | ||
|
|
a47b3cccb8 | ||
|
|
2070056d16 | ||
|
|
e227f1343f | ||
|
|
3c7335713d | ||
|
|
bdf4089264 | ||
|
|
b027113998 | ||
|
|
389c6b11dd | ||
|
|
fa5d9dc3b7 | ||
|
|
cb56728e57 | ||
|
|
b89c3a4573 | ||
|
|
0cc11108ba | ||
|
|
a7caf83022 | ||
|
|
053452c40c | ||
|
|
d4361c87ae | ||
|
|
6469eb7a0e | ||
|
|
13fbd0e802 | ||
|
|
e555a8f817 | ||
|
|
f9fb61cf1a | ||
|
|
43cf2e4e2c | ||
|
|
98f9a65202 | ||
|
|
8726c8fb73 | ||
|
|
27f3cb336f | ||
|
|
aa3bacd938 | ||
|
|
c71492ef32 | ||
|
|
70191f2d28 | ||
|
|
93db8b7ca7 | ||
|
|
d33b0cb9e3 | ||
|
|
0806d4ec25 | ||
|
|
9b646746b2 | ||
|
|
11993daec4 | ||
|
|
a78ffeeaba | ||
|
|
05b78339f6 | ||
|
|
365c221029 | ||
|
|
f60608a9c0 | ||
|
|
153d871be2 | ||
|
|
b69f2d7773 | ||
|
|
09ffe7ef6b | ||
|
|
3dfb94b524 | ||
|
|
d1e43d94e9 | ||
|
|
1b490e0e53 | ||
|
|
094146d630 | ||
|
|
23c2a53683 | ||
|
|
82b7689ca4 | ||
|
|
149f3e6f6d | ||
|
|
cea551c2ac | ||
|
|
2dcae23776 | ||
|
|
92e4e75217 | ||
|
|
5ca35bf77c | ||
|
|
c956b82d27 | ||
|
|
1c115096c4 | ||
|
|
c1d5fae018 | ||
|
|
17d49fc00f | ||
|
|
0e1e4c16b1 | ||
|
|
bec8e27b4f | ||
|
|
56841f0e50 | ||
|
|
723146050b | ||
|
|
85b1aa4c2d | ||
|
|
0506369519 | ||
|
|
ba1632974e | ||
|
|
c69082b1a4 | ||
|
|
74b2548982 | ||
|
|
f31656ec65 | ||
|
|
e91420c405 | ||
|
|
25df59a65d | ||
|
|
83fdd5720f | ||
|
|
db63241fd4 | ||
|
|
89b00c89aa | ||
|
|
d38917b5f0 | ||
|
|
910e0242c1 | ||
|
|
651b7bb75d | ||
|
|
c9f13ae1dd | ||
|
|
862e575100 | ||
|
|
4669c4541c | ||
|
|
769a8c41c4 | ||
|
|
bec9dba2b1 | ||
|
|
205ba2ea13 | ||
|
|
2073f6d287 | ||
|
|
0f25a960ee | ||
|
|
282ed3e309 | ||
|
|
cd031a7d38 | ||
|
|
e1885ed0bd | ||
|
|
4a31b619fa | ||
|
|
bd4464bd5e | ||
|
|
d907a7dc9f | ||
|
|
732070f750 | ||
|
|
48442b6b03 | ||
|
|
7fdbe547a3 | ||
|
|
59565b828d | ||
|
|
1e2d059890 | ||
|
|
0aa41908a2 | ||
|
|
b27ce3f79c | ||
|
|
238e52f74a | ||
|
|
9398b931fb | ||
|
|
b2a9785959 | ||
|
|
224a1f19a3 | ||
|
|
109c53f22b | ||
|
|
e25849b2cb | ||
|
|
57978accc1 | ||
|
|
ff37177f4d | ||
|
|
472d143021 | ||
|
|
157f95b08f | ||
|
|
6bf7ab0778 | ||
|
|
0de958be2a | ||
|
|
ef544fecf2 | ||
|
|
5eea68d6c6 | ||
|
|
5471367db1 | ||
|
|
b8265b1067 | ||
|
|
099c683a5a | ||
|
|
0357bb23e8 | ||
|
|
c7193b52fb | ||
|
|
ec14a65e23 | ||
|
|
0d70c6a0d0 | ||
|
|
bfa069c4d5 | ||
|
|
61bdf64e15 | ||
|
|
482b35c283 | ||
|
|
b74d886017 | ||
|
|
1667abad7e | ||
|
|
b187a853e7 | ||
|
|
228c7d142e | ||
|
|
041199644c | ||
|
|
3767f3633d | ||
|
|
af3253947e | ||
|
|
efc5eb2933 | ||
|
|
b4eeb96375 | ||
|
|
c9832e3d34 | ||
|
|
da3e3fc7a3 | ||
|
|
b5c83f0628 | ||
|
|
03087a55ba | ||
|
|
1ce3c16b30 | ||
|
|
bf702850a9 | ||
|
|
584c4cc05e | ||
|
|
279afd88bb | ||
|
|
c0a6d82025 | ||
|
|
3a03e1c93c | ||
|
|
bdaa70405f | ||
|
|
1281145982 | ||
|
|
87cac09477 | ||
|
|
5336129b58 | ||
|
|
72fc2b522d | ||
|
|
b6f6c84790 | ||
|
|
11e9be13b1 | ||
|
|
afdb8753ba | ||
|
|
f6a2e6739d | ||
|
|
d6569d510d | ||
|
|
04e4993d9b | ||
|
|
783e09d67d | ||
|
|
314f478225 | ||
|
|
c1dbc28aa2 | ||
|
|
8f7e393ffb | ||
|
|
b3055523b4 | ||
|
|
cf6b21564c | ||
|
|
996a4c023c | ||
|
|
0dcbdcc0e2 | ||
|
|
5bdd422db6 | ||
|
|
bf147f47b5 | ||
|
|
3f1f7faf34 | ||
|
|
1fc6725826 | ||
|
|
73958b9163 | ||
|
|
9b81a83894 | ||
|
|
5dee921300 | ||
|
|
c0dcf8925a |
No files matched your search
@@ -38,7 +38,12 @@ check_cxx_source_compiles(
|
||||
HAS_CLANG_PRESERVE_ALL)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
if (HAS_CLANG_PRESERVE_ALL)
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
if (MINGW_BUILD)
|
||||
message(STATUS "Ignoring broken clang::preserve_all support")
|
||||
set(HAS_CLANG_PRESERVE_ALL FALSE)
|
||||
else()
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
|
||||
@@ -652,7 +652,7 @@ def print_ir_allocator_helpers():
|
||||
|
||||
# Save NZCV if needed before clobbering NZCV
|
||||
if op.ImplicitFlagClobber:
|
||||
output_file.write("\t\tSaveNZCV();")
|
||||
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
|
||||
|
||||
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
|
||||
@@ -90,7 +90,6 @@ set (SRCS
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
@@ -101,9 +100,7 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
|
||||
@@ -321,16 +321,6 @@ namespace DefaultValues {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
{
|
||||
// Always fix up the number of threads and create the configuration
|
||||
// Otherwise the application could receive zero as the number of threads
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
@@ -31,15 +31,6 @@
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
|
||||
@@ -26,12 +26,6 @@ namespace FEXCore::Context {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CPUID.Init(this);
|
||||
return true;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
@@ -52,22 +46,10 @@ namespace FEXCore::Context {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
|
||||
return ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsDone() const {
|
||||
return IsPaused();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
|
||||
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
|
||||
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
@@ -14,8 +14,8 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -37,7 +37,6 @@
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
@@ -73,8 +72,6 @@ namespace FEXCore::Context {
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitializeContext() override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
@@ -92,15 +89,8 @@ namespace FEXCore::Context {
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
int GetProgramStatus() const override;
|
||||
|
||||
ExitReason GetExitReason() override;
|
||||
|
||||
bool IsDone() const override;
|
||||
|
||||
void GetCPUState(FEXCore::Core::CPUState *State) const override;
|
||||
void SetCPUState(const FEXCore::Core::CPUState *State) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
@@ -108,31 +98,37 @@ namespace FEXCore::Context {
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
@@ -290,17 +286,13 @@ namespace FEXCore::Context {
|
||||
void WaitForIdle() override;
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -311,7 +303,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
@@ -424,15 +416,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* InitCore and CreateThread both call this to finish up thread object initialization
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
@@ -450,7 +433,6 @@ namespace FEXCore::Context {
|
||||
|
||||
// Entry Cache
|
||||
std::mutex ExitMutex;
|
||||
fextl::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
@@ -28,7 +29,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
@@ -36,23 +37,23 @@ namespace x64 {
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
@@ -175,19 +176,20 @@ namespace x64 {
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
@@ -199,11 +201,10 @@ namespace x32 {
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
@@ -368,7 +369,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
@@ -399,6 +400,15 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
|
||||
movn(s, Reg.W(), (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
@@ -581,6 +591,7 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
// Disable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
@@ -592,10 +603,23 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
|
||||
// is always static and almost certainly clobbered by the subsequent code.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
|
||||
GPRSpillMask &= ~PFAFSpillMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
@@ -611,6 +635,14 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
|
||||
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
@@ -658,7 +690,7 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
bool FoundRegister{};
|
||||
[[maybe_unused]] bool FoundRegister{};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & GPRFillMask)) {
|
||||
TmpReg = Reg;
|
||||
@@ -688,6 +720,14 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
// is always static and was almost certainly clobbered.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
@@ -745,6 +785,11 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
|
||||
GPRFillMask &= ~PFAFMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
@@ -759,6 +804,14 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
|
||||
|
||||
@@ -53,6 +53,10 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
|
||||
|
||||
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
|
||||
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
|
||||
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
@@ -128,21 +132,21 @@ protected:
|
||||
|
||||
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
SpillForPreserveAllABICall(TMP1, true);
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
}
|
||||
else {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
}
|
||||
}
|
||||
|
||||
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(true);
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
}
|
||||
else {
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -771,6 +771,16 @@ public:
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
void axflag() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
void xaflag() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
// Conditional compare - register
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
|
||||
@@ -60,7 +60,7 @@ public:
|
||||
}
|
||||
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
|
||||
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Cryptographic two-register SHA
|
||||
|
||||
@@ -106,11 +106,10 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
PerCPUData.resize(Cores);
|
||||
|
||||
uint64_t MIDR{};
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec{};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
@@ -218,7 +217,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
fextl::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
if (MIDROption) {
|
||||
@@ -334,7 +333,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
}
|
||||
else {
|
||||
// If we aren't hybrid then just claim everything is big
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
|
||||
@@ -380,7 +379,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
// Processor Info and Features bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -389,7 +387,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(CoreCount << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
|
||||
Res.ecx =
|
||||
@@ -496,7 +494,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
|
||||
if (Leaf == 0) {
|
||||
// Report L1D
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Data | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
@@ -520,7 +518,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 1) {
|
||||
// Report L1I
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Instruction | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
@@ -544,7 +542,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
// Report L2
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b010 << 5) | // Cache level
|
||||
@@ -568,7 +566,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
@@ -1070,7 +1068,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
|
||||
(0 << 1) | // IRPerf: Instructions retired count support
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
Res.ecx =
|
||||
(0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
@@ -1168,7 +1166,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) con
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
@@ -1209,6 +1207,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
CTX = ctx;
|
||||
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
@@ -113,7 +113,7 @@ public:
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
uint32_t Cores{};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
|
||||
@@ -9,12 +9,11 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
@@ -46,6 +45,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "FEXCore/Utils/SignalScopeGuards.h"
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -76,64 +76,6 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
};
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
"PF",
|
||||
"",
|
||||
"AF",
|
||||
"",
|
||||
"ZF",
|
||||
"SF",
|
||||
"TF",
|
||||
"IF",
|
||||
"DF",
|
||||
"OF",
|
||||
"IOPL",
|
||||
"",
|
||||
"NT",
|
||||
"",
|
||||
"RF",
|
||||
"VM",
|
||||
"AC",
|
||||
"VIF",
|
||||
"VIP",
|
||||
"ID",
|
||||
};
|
||||
|
||||
std::string_view const& GetFlagName(unsigned Flag) {
|
||||
return FlagNames[Flag];
|
||||
}
|
||||
|
||||
constexpr std::array<std::string_view const, 16> RegNames = {
|
||||
"rax",
|
||||
"rbx",
|
||||
"rcx",
|
||||
"rdx",
|
||||
"rsi",
|
||||
"rdi",
|
||||
"rbp",
|
||||
"rsp",
|
||||
"r8",
|
||||
"r9",
|
||||
"r10",
|
||||
"r11",
|
||||
"r12",
|
||||
"r13",
|
||||
"r14",
|
||||
"r15",
|
||||
};
|
||||
|
||||
std::string_view const& GetGRegName(unsigned Reg) {
|
||||
return RegNames[Reg];
|
||||
}
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl()
|
||||
: IRCaptureCache {this} {
|
||||
@@ -157,6 +99,8 @@ namespace FEXCore::Context {
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
CPUID.Init(this);
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
@@ -221,7 +165,7 @@ namespace FEXCore::Context {
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) {
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
uint32_t EFLAGS{};
|
||||
|
||||
@@ -243,9 +187,23 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
uint32_t Packed_NZCV{};
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
if (WasInJIT) {
|
||||
// If we were in the JIT then NZCV is in the CPU's PSTATE object.
|
||||
// Packed in to the same bit locations as RFLAG_NZCV_LOC.
|
||||
Packed_NZCV = PSTATE;
|
||||
|
||||
// If we were in the JIT then PF and AF are in registers.
|
||||
// Move them to the CPUState frame now.
|
||||
Frame->State.pf_raw = HostGPRs[CPU::REG_PF.Idx()];
|
||||
Frame->State.af_raw = HostGPRs[CPU::REG_AF.Idx()];
|
||||
}
|
||||
else {
|
||||
// If we were not in the JIT then the NZCV state is stored in the CPUState RFLAG_NZCV_LOC.
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
}
|
||||
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
@@ -259,13 +217,13 @@ namespace FEXCore::Context {
|
||||
|
||||
// PF calculation is deferred, calculate it now.
|
||||
// Popcount the 8-bit flag and then extract the lower bit.
|
||||
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_RAW_LOC];
|
||||
uint32_t PFByte = Frame->State.pf_raw & 0xff;
|
||||
uint32_t PF = std::popcount(PFByte ^ 1) & 1;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_RAW_LOC;
|
||||
|
||||
// AF calculation is deferred, calculate it now.
|
||||
// XOR with PF byte and extract bit 4.
|
||||
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_RAW_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
return EFLAGS;
|
||||
@@ -285,11 +243,11 @@ namespace FEXCore::Context {
|
||||
// AF stored in bit 4 in our internal representation. It is also
|
||||
// XORed with byte 4 of the PF byte, but we write that as zero here so
|
||||
// we don't need any special handling for that.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
Frame->State.af_raw = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
break;
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
// PF is inverted in our internal representation.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
Frame->State.pf_raw = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
break;
|
||||
default:
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 1 : 0;
|
||||
@@ -359,13 +317,6 @@ namespace FEXCore::Context {
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
else {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
@@ -373,36 +324,21 @@ namespace FEXCore::Context {
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
// FEX needs to start paused when gdb is enabled.
|
||||
StartPaused = true;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(InitialRIP, StackPointer, nullptr, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
InitializeThreadData(Thread);
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::StartGdbServer() {
|
||||
#ifndef _WIN32
|
||||
if (!DebugServer) {
|
||||
DebugServer = fextl::make_unique<GdbServer>(this, SignalDelegation, SyscallHandler);
|
||||
StartPaused = true;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::StopGdbServer() {
|
||||
#ifndef _WIN32
|
||||
DebugServer.reset();
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
@@ -575,14 +511,6 @@ namespace FEXCore::Context {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
}
|
||||
|
||||
int ContextImpl::GetProgramStatus() const {
|
||||
return ParentThread->StatusCode;
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CPUBackend->Initialize();
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
ContextImpl *This;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
@@ -681,20 +609,22 @@ namespace FEXCore::Context {
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
Thread->ThreadManager.parent_tid = ParentTID;
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
@@ -1113,7 +1043,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1296,7 +1226,7 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
@@ -1305,7 +1235,7 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1333,7 +1263,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
@@ -1381,17 +1311,11 @@ namespace FEXCore::Context {
|
||||
|
||||
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
|
||||
@@ -85,7 +85,9 @@ public:
|
||||
#endif
|
||||
|
||||
uint16_t GetSRAGPRCount() const {
|
||||
return StaticRegisters.size();
|
||||
// PF/AF are the final two SRA registers.
|
||||
// Only return the SRA for GPRs.
|
||||
return StaticRegisters.size() - 2;
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const {
|
||||
@@ -93,7 +95,7 @@ public:
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,11 +180,13 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
if (EnableCrypto) {
|
||||
Features->SupportsAES = true;
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsSHA = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
}
|
||||
else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsSHA = false;
|
||||
Features->SupportsPMULL_128Bit = false;
|
||||
}
|
||||
if (EnableRPRES) {
|
||||
@@ -211,6 +213,8 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) &&
|
||||
Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
@@ -238,13 +242,14 @@ HostFeatures::HostFeatures() {
|
||||
#endif
|
||||
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
|
||||
SupportsAVX2 = false;
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
|
||||
SupportsAFP = false;
|
||||
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
|
||||
SupportsRPRES = false;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
@@ -283,6 +288,8 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
// Simulator doesn't support SHA
|
||||
SupportsSHA = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
|
||||
@@ -87,7 +87,7 @@ DEF_OP(Add) {
|
||||
|
||||
DEF_OP(AddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AddNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
@@ -98,78 +98,63 @@ DEF_OP(AddNZCV) {
|
||||
} else {
|
||||
cmn(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(AdcNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AdcNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
// TODO: Optimize this out
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->NZCV.ID()));
|
||||
|
||||
adcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(SbbNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SbbNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
// Carry-in needs to be inverted for subtractions due to carry versus borrow
|
||||
// distinction between x86 and arm.
|
||||
// See below remarks on cfinv
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, GetReg(Op->NZCV.ID()), 1u << 29);
|
||||
|
||||
// TODO: Optimize this out
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
sbcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// The carry flag produced by arm64 sbcs is inverted compared to the x86 carry
|
||||
// flag. Invert it now.
|
||||
//
|
||||
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
|
||||
// that's only available with Feat_FlagM. For now the portable way is to flip
|
||||
// bit 29 (carry) manually.
|
||||
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src1.ID());
|
||||
uint64_t Const;
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// setf+rmif would avoid the scratch register, but higher latency on M1.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
// zeroing CV.
|
||||
if (OpSize < 4) {
|
||||
lsl(EmitSize, Dst, Src, 32 - (OpSize * 8));
|
||||
Src = Dst;
|
||||
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
|
||||
// needed.
|
||||
if (Op->Src1 != Op->Src2) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, TMP1, Src1, Const);
|
||||
} else {
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
}
|
||||
|
||||
Src1 = TMP1;
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
tst(EmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
@@ -199,7 +184,7 @@ DEF_OP(SubShift) {
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
@@ -213,20 +198,70 @@ DEF_OP(SubNZCV) {
|
||||
} else {
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
DEF_OP(CarryInvert) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
cfinv();
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
DEF_OP(RmifNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_RmifNZCV>();
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
if (Op->InvertCarry) {
|
||||
// The carry flag produced by arm64 subs is inverted compared to the x86 carry
|
||||
// flag. Invert it now.
|
||||
//
|
||||
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
|
||||
// that's only available with Feat_FlagM. For now the portable way is to flip
|
||||
// bit 29 (carry) manually.
|
||||
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
|
||||
rmif(GetReg(Op->Src.ID()).X(), Op->Rotate, Op->Mask);
|
||||
}
|
||||
|
||||
DEF_OP(AXFlag) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
|
||||
axflag();
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
|
||||
uint64_t Const = 0;
|
||||
auto Src1 = IsInlineConstant(Op->Src1, &Const) ? ARMEmitter::Reg::zr :
|
||||
GetReg(Op->Src1.ID());
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "Unsupported inline constant");
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmn(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
|
||||
} else {
|
||||
ccmn(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -237,27 +272,10 @@ DEF_OP(Neg) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
|
||||
abs(EmitSize, Dst, Src);
|
||||
}
|
||||
else {
|
||||
cmp(EmitSize, Src, 0);
|
||||
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
if (Op->Cond == FEXCore::IR::COND_AL)
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
else
|
||||
cneg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()), MapSelectCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
@@ -559,6 +577,16 @@ DEF_OP(Xor) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1307,36 +1335,6 @@ DEF_OP(Sbfe) {
|
||||
sbfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1345,28 +1343,15 @@ DEF_OP(Select) {
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
LOGMAN_THROW_A_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
|
||||
if (tests) {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
tst(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
tst(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
@@ -1405,6 +1390,38 @@ DEF_OP(Select) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelect) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
if (is_const_true) {
|
||||
if (is_const_false != true || !(const_true == 1 || const_true == all_ones) || const_false != 0) {
|
||||
LOGMAN_MSG_A_FMT("NZCVSelect: Unsupported constant");
|
||||
}
|
||||
|
||||
if (const_true == all_ones)
|
||||
csetm(EmitSize, Dst, cc);
|
||||
else
|
||||
cset(EmitSize, Dst, cc);
|
||||
} else if (is_const_false) {
|
||||
LOGMAN_THROW_A_FMT(const_false == 0, "NZCVSelect: unsupported constant");
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), ARMEmitter::Reg::zr, cc);
|
||||
} else {
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetReg(Op->FalseVal.ID()), cc);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1519,41 +1536,10 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
const auto EmitSubSize = Op->ElementSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
|
||||
|
||||
fcmp(EmitSubSize, Scalar1, Scalar2);
|
||||
bool set = false;
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
|
||||
LOGMAN_THROW_AA_FMT(IR::FCMP_FLAG_EQ == 0, "IR::FCMP_FLAG_EQ must equal 0");
|
||||
// EQ or unordered
|
||||
cset(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Condition::CC_EQ); // Z = 1
|
||||
csinc(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_VC); // IF !V ? Z : 1
|
||||
set = true;
|
||||
}
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
// LT or unordered
|
||||
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_LT);
|
||||
if (!set) {
|
||||
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT);
|
||||
set = true;
|
||||
} else {
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT, 1);
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
|
||||
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_VS);
|
||||
if (!set) {
|
||||
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED);
|
||||
set = true;
|
||||
} else {
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -97,9 +97,7 @@ DEF_OP(Jump) {
|
||||
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
@@ -117,8 +115,8 @@ static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
@@ -130,42 +128,27 @@ DEF_OP(CondJump) {
|
||||
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
if (Op->FromNZCV) {
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (tests) {
|
||||
if (isConst) {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
} else {
|
||||
if (isConst) {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ ||
|
||||
Op->Cond.Val == FEXCore::IR::COND_NEQ,
|
||||
"CondJump: Expected simple condition");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
}
|
||||
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
// TODO: Wire up tbz/tbnz
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
|
||||
@@ -179,6 +179,32 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256su0(Dst, Src2);
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256su0(VTMP1, Src2);
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -5,6 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
@@ -295,7 +296,11 @@ DEF_OP(LoadRegisterSRA) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regId =
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
@@ -473,10 +478,14 @@ DEF_OP(StoreRegisterSRA) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
const auto regId =
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
@@ -1031,23 +1040,37 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadNZCV) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(StoreNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_StoreNZCV>();
|
||||
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
|
||||
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
|
||||
"PF/AF must be accessed as registers");
|
||||
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
|
||||
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
|
||||
"PF/AF must be accessed as registers");
|
||||
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
|
||||
@@ -1840,9 +1863,15 @@ DEF_OP(MemSet) {
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
// If Direction == 0 then:
|
||||
// MemReg is incremented (by size)
|
||||
// else:
|
||||
@@ -1862,8 +1891,10 @@ DEF_OP(MemSet) {
|
||||
add(TMP2, Prefix.X(), MemReg.X());
|
||||
}
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
@@ -1917,8 +1948,7 @@ DEF_OP(MemSet) {
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
auto EmitMemset = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
@@ -1978,15 +2008,26 @@ DEF_OP(MemSet) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
if (DirectionIsInline) {
|
||||
// If the direction constant is set then the direction is negative.
|
||||
EmitMemset(DirectionConstant ? -1 : 1);
|
||||
}
|
||||
else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
EmitMemset(Direction);
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
@@ -2001,7 +2042,12 @@ DEF_OP(MemCpy) {
|
||||
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
auto Dst = GetRegPair(Node);
|
||||
// If Direction == 0 then:
|
||||
@@ -2038,8 +2084,10 @@ DEF_OP(MemCpy) {
|
||||
// TMP3 = Src
|
||||
// TMP4 = load+store temp value
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
@@ -2161,8 +2209,7 @@ DEF_OP(MemCpy) {
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
auto EmitMemcpy = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
@@ -2235,15 +2282,24 @@ DEF_OP(MemCpy) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
if (DirectionIsInline) {
|
||||
// If the direction constant is set then the direction is negative.
|
||||
EmitMemcpy(DirectionConstant ? -1 : 1);
|
||||
}
|
||||
else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
|
||||
@@ -1799,6 +1799,9 @@ DEF_OP(VFMin) {
|
||||
}
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2: {
|
||||
fcmp(Vector1.H(), Vector2.H());
|
||||
@@ -1818,6 +1821,9 @@ DEF_OP(VFMin) {
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on false.
|
||||
@@ -1878,6 +1884,9 @@ DEF_OP(VFMax) {
|
||||
}
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2: {
|
||||
fcmp(Vector1.H(), Vector2.H());
|
||||
@@ -1897,6 +1906,9 @@ DEF_OP(VFMax) {
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on true.
|
||||
@@ -2654,6 +2666,9 @@ DEF_OP(VCMPEQ) {
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
@@ -2664,6 +2679,9 @@ DEF_OP(VCMPEQ) {
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmeq(SubRegSize.Scalar, Dst, Vector1, Vector2);
|
||||
@@ -2695,6 +2713,9 @@ DEF_OP(VCMPEQZ) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Ensure no junk is in the temp (important for ensuring
|
||||
// non-equal entries remain as zero).
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
|
||||
@@ -2705,6 +2726,9 @@ DEF_OP(VCMPEQZ) {
|
||||
cmpeq(SubRegSize.Vector, ComparePred, Mask, Vector.Z(), 0);
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmeq(SubRegSize.Scalar, Dst, Vector);
|
||||
@@ -2737,6 +2761,9 @@ DEF_OP(VCMPGT) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// General idea is to compare for greater-than, bitwise NOT
|
||||
// the valid values, then ORR the NOTed values with the original
|
||||
// values to form entries that are all 1s.
|
||||
@@ -2744,6 +2771,9 @@ DEF_OP(VCMPGT) {
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmgt(SubRegSize.Scalar, Dst, Vector1, Vector2);
|
||||
@@ -2775,6 +2805,9 @@ DEF_OP(VCMPGTZ) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Ensure no junk is in the temp (important for ensuring
|
||||
// non greater-than values remain as zero).
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
|
||||
@@ -2782,6 +2815,9 @@ DEF_OP(VCMPGTZ) {
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmgt(SubRegSize.Scalar, Dst, Vector);
|
||||
@@ -2813,6 +2849,9 @@ DEF_OP(VCMPLTZ) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Ensure no junk is in the temp (important for ensuring
|
||||
// non less-than values remain as zero).
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
|
||||
@@ -2820,6 +2859,9 @@ DEF_OP(VCMPLTZ) {
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector.Z());
|
||||
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmlt(SubRegSize.Scalar, Dst, Vector);
|
||||
@@ -3904,6 +3946,58 @@ DEF_OP(VUShrI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUShraI) {
|
||||
const auto Op = IROp->C<IR::IROp_VUShraI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst == DestVector) {
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
}
|
||||
else {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), DestVector.Z());
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.Z(), DestVector.Z());
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (Dst == DestVector) {
|
||||
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
|
||||
}
|
||||
else {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Q(), DestVector.Q());
|
||||
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.Q(), DestVector.Q());
|
||||
usra(SubRegSize, VTMP1.Q(), Vector.Q(), BitShift);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSShrI) {
|
||||
const auto Op = IROp->C<IR::IROp_VSShrI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -38,6 +38,13 @@ enum class MemoryAccessType {
|
||||
STREAM,
|
||||
};
|
||||
|
||||
enum class BTAction {
|
||||
BTNone,
|
||||
BTClear,
|
||||
BTSet,
|
||||
BTComplement,
|
||||
};
|
||||
|
||||
struct LoadSourceOptions {
|
||||
// Alignment of the load in bytes. -1 signifies unaligned
|
||||
int8_t Align = -1;
|
||||
@@ -89,7 +96,6 @@ public:
|
||||
TYPE_RORI,
|
||||
TYPE_ROL,
|
||||
TYPE_ROLI,
|
||||
TYPE_FCMP,
|
||||
TYPE_BEXTR,
|
||||
TYPE_BLSI,
|
||||
TYPE_BLSMSK,
|
||||
@@ -98,7 +104,6 @@ public:
|
||||
TYPE_BZHI,
|
||||
TYPE_TZCNT,
|
||||
TYPE_LZCNT,
|
||||
TYPE_BITSELECT,
|
||||
TYPE_RDRAND,
|
||||
};
|
||||
|
||||
@@ -122,11 +127,10 @@ public:
|
||||
}
|
||||
|
||||
void StartNewBlock() {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
|
||||
// If we loaded flags but didn't change them, invalidate the cached copy and move on.
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
PossiblySetNZCVBits = ~0U;
|
||||
|
||||
// New block needs to reset segment telemetry.
|
||||
SegmentsNeedReadCheck = ~0U;
|
||||
@@ -155,6 +159,14 @@ public:
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, ssa1, ssa2, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// The jump will ignore the sources, so it doesn't matter what we put here.
|
||||
// Put an inline constant so RA+codegen will ignore altogether.
|
||||
auto Placeholder = _InlineConstant(0);
|
||||
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
// If we are switching to a new block and this current block has yet to set a RIP
|
||||
@@ -324,14 +336,10 @@ public:
|
||||
void RCLOp1Bit(OpcodeArgs);
|
||||
void RCLOp(OpcodeArgs);
|
||||
void RCLSmallerOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
|
||||
template<uint32_t SrcIndex, enum BTAction Action>
|
||||
void BTOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void BTROp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void BTSOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void BTCOp(OpcodeArgs);
|
||||
|
||||
void IMUL1SrcOp(OpcodeArgs);
|
||||
void IMUL2SrcOp(OpcodeArgs);
|
||||
void IMULOp(OpcodeArgs);
|
||||
@@ -916,17 +924,36 @@ public:
|
||||
}
|
||||
|
||||
protected:
|
||||
void SaveNZCV() override {
|
||||
void SaveNZCV(IROps Op = OP_DUMMY) override {
|
||||
/* Some opcodes are conservatively marked as clobbering flags, but in fact
|
||||
* do not clobber flags in certain conditions. Check for that here as an
|
||||
* optimization.
|
||||
*/
|
||||
switch (Op) {
|
||||
case OP_VFMINSCALARINSERT:
|
||||
case OP_VFMAXSCALARINSERT:
|
||||
/* On AFP platforms, becomes fmin/fmax and preserves NZCV. Otherwise
|
||||
* becomes fcmp and clobbers.
|
||||
*/
|
||||
if (CTX->HostFeatures.SupportsAFP)
|
||||
return;
|
||||
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
// Invariant: When executing instructions that clobber NZCV, the flags must
|
||||
// be resident in a GPR, which is equivalent to CachedNZCV != nullptr. Get
|
||||
// the NZCV which fills the cache if necessary.
|
||||
if (CachedNZCV == nullptr)
|
||||
GetNZCV();
|
||||
|
||||
// Assume we'll need a reload.
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
@@ -934,13 +961,6 @@ private:
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
constexpr static unsigned FullNZCVMask =
|
||||
(1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) |
|
||||
@@ -1243,7 +1263,7 @@ private:
|
||||
|
||||
OrderedNode *GetNZCV() {
|
||||
if (!CachedNZCV) {
|
||||
CachedNZCV = _LoadFlag(FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
CachedNZCV = _LoadNZCV();
|
||||
|
||||
// We don't know what's set
|
||||
PossiblySetNZCVBits = ~0;
|
||||
@@ -1275,37 +1295,85 @@ private:
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
CachedNZCV = _LoadNZCV();
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
NZCVDirty = true;
|
||||
NZCVDirty = false;
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
unsigned Bit = IndexNZCV(BitOffset);
|
||||
void InsertNZCV(unsigned BitOffset, OrderedNode *Value, signed FlagOffset, bool MustMask) {
|
||||
signed Bit = IndexNZCV(BitOffset);
|
||||
|
||||
// If NZCV is not dirty, we always want to use rmif, it's 1 instruction to
|
||||
// implement this. But if NZCV is dirty, it might still be cheaper to copy
|
||||
// the GPR flags to NZCV and rmif. This is a heuristic for cases where we
|
||||
// expect that 2 instruction sequence to be a win (versus something like
|
||||
// bfe+mov+bfi+mov which can happen with our RA..). It's not totally
|
||||
// conservative but it's pretty good in practice.
|
||||
bool PreferRmif = !NZCVDirty || FlagOffset || MustMask ||
|
||||
(PossiblySetNZCVBits & (1u << Bit));
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM && PreferRmif) {
|
||||
// Update NZCV
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
_StoreNZCV(CachedNZCV);
|
||||
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
|
||||
// Insert as NZCV.
|
||||
signed RmifBit = Bit - 28;
|
||||
_RmifNZCV(Value, (64 + FlagOffset - RmifBit) % 64, 1u << RmifBit);
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Insert as GPR
|
||||
if (FlagOffset || MustMask)
|
||||
Value = _Bfe(OpSize::i64Bit, 1, FlagOffset, Value);
|
||||
|
||||
if (PossiblySetNZCVBits == 0)
|
||||
SetNZCV(_Lshl(OpSize::i64Bit, Value, _Constant(Bit)));
|
||||
else if ((PossiblySetNZCVBits & (1u << Bit)) == 0)
|
||||
SetNZCV(_Orlshl(OpSize::i32Bit, GetNZCV(), Value, Bit));
|
||||
else
|
||||
SetNZCV(_Bfi(OpSize::i32Bit, 1, Bit, GetNZCV(), Value));
|
||||
}
|
||||
|
||||
uint32_t SetBits = PossiblySetNZCVBits;
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
}
|
||||
|
||||
if (SetBits == 0)
|
||||
return _Lshl(OpSize::i64Bit, Value, _Constant(Bit));
|
||||
else if ((SetBits & (1u << Bit)) == 0)
|
||||
return _Orlshl(OpSize::i32Bit, NZCV, Value, Bit);
|
||||
else
|
||||
return _Bfi(OpSize::i32Bit, 1, Bit, NZCV, Value);
|
||||
void CarryInvert() {
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
|
||||
// Invert as NZCV.
|
||||
_CarryInvert();
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits |= 1u << Bit;
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value) {
|
||||
SetRFLAG(Value, BitOffset);
|
||||
void SetRFLAG(OrderedNode *Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, BitOffset, ValueOffset, MustMask);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else {
|
||||
if (ValueOffset || MustMask)
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
|
||||
if (IsNZCV(BitOffset))
|
||||
SetNZCV(InsertNZCV(PossiblySetNZCVBits ? GetNZCV() : nullptr, BitOffset, Value));
|
||||
else
|
||||
_StoreFlag(Value, BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
void SetAF(unsigned Constant) {
|
||||
@@ -1318,17 +1386,134 @@ private:
|
||||
|
||||
void ZeroMultipleFlags(uint32_t BitMask);
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset) {
|
||||
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_PL} : CondClassType{COND_MI};
|
||||
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_NEQ} : CondClassType{COND_EQ};
|
||||
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_ULT} : CondClassType{COND_UGE};
|
||||
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_FNU} : CondClassType{COND_FU};
|
||||
|
||||
default:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset, bool Invert = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
if (!CachedNZCV || (PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset))))
|
||||
return _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
else
|
||||
return _Constant(0);
|
||||
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
|
||||
return _Constant(Invert ? 1 : 0);
|
||||
} else if (NZCVDirty) {
|
||||
auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
|
||||
if (Invert)
|
||||
return _Xor(OpSize::i32Bit, Value, _Constant(1));
|
||||
else
|
||||
return Value;
|
||||
} else {
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert),
|
||||
_Constant(1), _Constant(0));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else {
|
||||
return _LoadFlag(BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
// Set SSE comparison flags based on the result set by Arm FCMP. This converts
|
||||
// NZCV from the Arm representation to an eXternal representation that's
|
||||
// totally not a euphemism for x86 or anything, nuh-uh.
|
||||
void ConvertNZCVToSSE() {
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
|
||||
// after. We can recover "unordered" after axflag as (Z && !C), but
|
||||
// there's no condition code for this so it would take 2 instructions
|
||||
// instead of one, which seems worse than doing 1 op before and breaking
|
||||
// the fusion.
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
OrderedNode *PFInvert =
|
||||
_NZCVSelect(OpSize::i32Bit, CondClassType{COND_FNU}, _Constant(1), _Constant(0));
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PFInvert);
|
||||
|
||||
// For the rest, this one weird a64 instruction maps exactly to what x86
|
||||
// needs. What a coincidence!
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// It does assume we invert CF internally, which is still TODO for us. For
|
||||
// now, add a cfinv to deal. Hopefully we delete this later.
|
||||
CarryInvert();
|
||||
} else {
|
||||
OrderedNode *Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
OrderedNode *C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
OrderedNode *V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
// We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do
|
||||
// this all with shifted-or's on non-flagm platforms.
|
||||
ZeroNZCV();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Or(OpSize::i32Bit, C_inv, V));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
|
||||
// Note that we store PF inverted.
|
||||
// TODO: We could maybe optimize this xor out for non-flagm platforms with
|
||||
// bfi/bfxil?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Xor(OpSize::i32Bit, V, _Constant(1)));
|
||||
}
|
||||
}
|
||||
|
||||
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
|
||||
// NZCV on flagm2 platforms.
|
||||
void ConvertNZCVToX87() {
|
||||
OrderedNode *V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// Convert to x86 flags, saves us from or'ing after.
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// Copy the values. CF is inverted from the axflag result, ZF is as-is.
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
} else {
|
||||
OrderedNode *Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
OrderedNode *N = GetRFLAG(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Or(OpSize::i32Bit, N, V));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
}
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(V);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
// replaced with NewOp. Useful for generic building code. Not safe in general.
|
||||
// but does the right handling of ImplicitFlagClobber at least and must be
|
||||
// used instead of raw Op mutation.
|
||||
#define DeriveOp(Dest, NewOp, Expr) \
|
||||
if (ImplicitFlagClobber(NewOp)) \
|
||||
SaveNZCV(NewOp); \
|
||||
auto Dest = (Expr); \
|
||||
Dest.first->Header.Op = (NewOp)
|
||||
|
||||
// Named constant cache for the current block.
|
||||
// Different arrays for sizes 1,2,4,8,16,32.
|
||||
OrderedNode *CachedNamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MAX][6]{};
|
||||
@@ -1382,8 +1567,8 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
OrderedNode *SelectMask(OrderedNode *Cmp, uint64_t Mask, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
OrderedNode *SelectNZCV(unsigned BitOffset, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
|
||||
OrderedNode *SelectBit(OrderedNode *Cmp, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
OrderedNode *SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
|
||||
/**
|
||||
@@ -1416,7 +1601,7 @@ private:
|
||||
OrderedNode *Res{};
|
||||
|
||||
union {
|
||||
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT, RDRAND
|
||||
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, RDRAND
|
||||
struct {
|
||||
} NoSource;
|
||||
|
||||
@@ -1486,13 +1671,48 @@ private:
|
||||
return CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
template <typename F>
|
||||
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
|
||||
// We are the ones calculating the deferred flags. Don't recurse!
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
// RCR can call this with constants, so handle that without branching.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Shift), &Const)) {
|
||||
if (Const)
|
||||
CalculateFlags();
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
// Otherwise, prepare to branch.
|
||||
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
// If the shift is zero, do not touch the flags.
|
||||
auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto EndBlock = CreateNewCodeBlockAfter(SetBlock);
|
||||
CondJump(Shift, Zero, EndBlock, SetBlock, {COND_EQ});
|
||||
|
||||
SetCurrentCodeBlock(SetBlock);
|
||||
StartNewBlock();
|
||||
{
|
||||
CalculateFlags();
|
||||
Jump(EndBlock);
|
||||
}
|
||||
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
StartNewBlock();
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
}
|
||||
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
OrderedNode *LoadPFRaw();
|
||||
OrderedNode *LoadAF();
|
||||
void FixupAF();
|
||||
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
void CalculatePF(OrderedNode *Res);
|
||||
void CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
void CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool Sub);
|
||||
@@ -1515,7 +1735,6 @@ private:
|
||||
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BLSMSK(OrderedNode *Src);
|
||||
@@ -1524,7 +1743,6 @@ private:
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculateFlags_RDRAND(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
@@ -1827,20 +2045,6 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_FCMP(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_FCMP,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BEXTR,
|
||||
@@ -1915,14 +2119,6 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BITSELECT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BITSELECT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_RDRAND,
|
||||
|
||||
@@ -26,9 +26,24 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Tmp = _Ror(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
|
||||
auto Top = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Src, 3), Tmp);
|
||||
auto Result = _VInsGPR(16, 4, 3, Src, Top);
|
||||
OrderedNode *RotatedNode{};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
// Move the element to zero, rotate, and then move back (Using duplicates).
|
||||
// Saves one instruction versus that path that doesn't support SHA extension.
|
||||
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Sha1HRotated = _VSha1H(Duplicated);
|
||||
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
|
||||
}
|
||||
else {
|
||||
// SHA1 extension missing, manually rotate.
|
||||
// Emulate rotate.
|
||||
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
|
||||
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
|
||||
}
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -49,23 +64,31 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ROR by 31 is equivalent to a ROL by 1
|
||||
auto ThirtyOne = _Constant(32, 31);
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
|
||||
auto W13 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W16 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), W13), ThirtyOne);
|
||||
auto W17 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), W14), ThirtyOne);
|
||||
auto W18 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), W15), ThirtyOne);
|
||||
auto W19 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), W16), ThirtyOne);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W16);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W17);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W18);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W19);
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -151,30 +174,37 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
OrderedNode *Result{};
|
||||
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
}
|
||||
else {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
Result = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
@@ -40,7 +40,6 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
};
|
||||
|
||||
void OpDispatchBuilder::ZeroMultipleFlags(uint32_t FlagsMask) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
if (ContainsNZCV(FlagsMask)) {
|
||||
@@ -130,8 +129,7 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
} else {
|
||||
auto Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
SetRFLAG(Src, FlagOffset, FlagOffset, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -237,17 +235,17 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNo
|
||||
Anded = _Andn(OpSize, XorOp2, XorOp1);
|
||||
}
|
||||
|
||||
auto OF = _Bfe(OpSize, 1, SrcSize * 8 - 1, Anded);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
// Read the stored byte. This is the original 8-bit result, it needs parity calculated.
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
// Read the stored byte. This is the original result (up to 64-bits), it needs
|
||||
// parity calculated.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
|
||||
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
|
||||
auto InputFPR = _VCastFromGPR(4, 4, PFByte);
|
||||
auto InputFPR = _VCastFromGPR(4, 4, Result);
|
||||
|
||||
// Calculate the popcount.
|
||||
auto Count = _VPopcount(1, 1, InputFPR);
|
||||
@@ -255,14 +253,14 @@ OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadAF() {
|
||||
// Read the stored byte. This is the XOR of the arguments.
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
// Read the stored value. This is the XOR of the arguments.
|
||||
auto AFWord = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Read the result, stored as the PF byte for deferred PF calculation.
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
// Read the result, stored for PF.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// What's left is to XOR and extract. This is the deferred part.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFByte, PFByte));
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFWord, Result));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FixupAF() {
|
||||
@@ -272,23 +270,14 @@ void OpDispatchBuilder::FixupAF() {
|
||||
//
|
||||
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
|
||||
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFByte, PFByte);
|
||||
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
|
||||
// For shifts, we can only update for nonzero shift. If zero, we nop out the flag write by
|
||||
// writing the existing value. Note we call GetRFLAG directly, rather than LoadPFRaw, because
|
||||
// we need the existing /encoded/ value rather than the decoded PF value. In particular,
|
||||
// this does not calculate a popcount.
|
||||
if (condition) {
|
||||
auto OldFlag = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
Res = _Select(FEXCore::IR::COND_EQ, condition, _Constant(0), OldFlag, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode *Res) {
|
||||
// Calculation is entirely deferred until load, just store the 8-bit result.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
|
||||
}
|
||||
@@ -313,7 +302,7 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
if (CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE) {
|
||||
// Nothing to do
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
_StoreFlag(CachedNZCV, FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
_StoreNZCV(CachedNZCV);
|
||||
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
@@ -446,13 +435,6 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_FCMP:
|
||||
CalculateFlags_FCMP(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BEXTR:
|
||||
CalculateFlags_BEXTR(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
@@ -487,9 +469,6 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BITSELECT:
|
||||
CalculateFlags_BITSELECT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_RDRAND:
|
||||
CalculateFlags_RDRAND(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
@@ -501,7 +480,7 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
|
||||
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
_StoreFlag(CachedNZCV, FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
_StoreNZCV(CachedNZCV);
|
||||
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
@@ -516,7 +495,12 @@ void OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
CalculatePF(Res);
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
SetNZCV(_AdcNZCV(OpSize, Src1, Src2, GetNZCV()));
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
_StoreNZCV(CachedNZCV);
|
||||
CachedNZCV = nullptr;
|
||||
|
||||
_AdcNZCV(OpSize, Src1, Src2);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
} else {
|
||||
// SF/ZF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -544,7 +528,19 @@ void OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
CalculatePF(Res);
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
SetNZCV(_SbbNZCV(OpSize, Src1, Src2, GetNZCV()));
|
||||
// Rectify input carry
|
||||
CarryInvert();
|
||||
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
_StoreNZCV(CachedNZCV);
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
|
||||
_SbbNZCV(OpSize, Src1, Src2);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// Rectify output carry
|
||||
CarryInvert();
|
||||
} else {
|
||||
// SF/ZF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -574,8 +570,14 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
|
||||
// TODO: Could do this path for small sources if we have FEAT_FlagM
|
||||
if (SrcSize >= 4) {
|
||||
_SubNZCV(OpSize, Src1, Src2);
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// We only bother inverting CF if we're actually going to update CF.
|
||||
SetNZCV(_SubNZCV(OpSize, Src1, Src2, UpdateCF));
|
||||
if (UpdateCF)
|
||||
CarryInvert();
|
||||
} else {
|
||||
// SF/ZF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -583,8 +585,7 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
// CF
|
||||
if (UpdateCF) {
|
||||
// Grab carry bit from unmasked output.
|
||||
auto Bfe = _Bfe(OpSize::i32Bit, 1, SrcSize * 8, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Bfe);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SrcSize * 8, true);
|
||||
}
|
||||
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, true);
|
||||
@@ -606,7 +607,10 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
|
||||
// TODO: Could do this path for small sources if we have FEAT_FlagM
|
||||
if (SrcSize >= 4) {
|
||||
SetNZCV(_AddNZCV(OpSize, Src1, Src2));
|
||||
_AddNZCV(OpSize, Src1, Src2);
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
PossiblySetNZCVBits = ~0;
|
||||
} else {
|
||||
// SF/ZF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -614,8 +618,7 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
// CF
|
||||
if (UpdateCF) {
|
||||
// Grab carry bit from unmasked output
|
||||
auto Bfe = _Bfe(OpSize::i32Bit, 1, SrcSize * 8, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Bfe);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SrcSize * 8, true);
|
||||
}
|
||||
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, false);
|
||||
@@ -627,8 +630,6 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
// PF/AF/ZF/SF
|
||||
// Undefined
|
||||
{
|
||||
@@ -640,19 +641,22 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
{
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC)));
|
||||
|
||||
// Set CV accordingly and zero NZ regardless
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, High, SignBit, Zero, CV));
|
||||
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
|
||||
// undefined, this does what we need.
|
||||
auto Zero = _Constant(0);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
PossiblySetNZCVBits = ~0;
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
auto Zero = _Constant(0);
|
||||
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
|
||||
|
||||
// AF/SF/PF/ZF
|
||||
// Undefined
|
||||
@@ -665,11 +669,14 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
{
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
_SubNZCV(Size, High, Zero);
|
||||
|
||||
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC)));
|
||||
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, High, Zero, Zero, CV));
|
||||
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
PossiblySetNZCVBits = ~0;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -685,118 +692,79 @@ void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
auto OldNZCV = GetNZCV();
|
||||
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
auto Size = _Constant(SrcSize * 8);
|
||||
auto ShiftAmt = _Sub(OpSize, Size, Src2);
|
||||
auto LastBit = _Bfe(OpSize, 1, 0, _Lshr(OpSize, Src1, ShiftAmt));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
|
||||
}
|
||||
auto LastBit = _Lshr(OpSize, Src1, ShiftAmt);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
|
||||
|
||||
CalculatePF(Res, Src2);
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// OF
|
||||
{
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
// When Shift > 1 then OF is undefined
|
||||
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
|
||||
}
|
||||
|
||||
// Now select between the two
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
auto OFXor = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OFXor, SrcSize * 8 - 1, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
auto OldNZCV = GetNZCV();
|
||||
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, One);
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, _Constant(1));
|
||||
const auto CFSize = IR::SizeToOpSize(std::max<uint8_t>(4u, SrcSize));
|
||||
auto LastBit = _Bfe(CFSize, 1, 0, _Lshr(CFSize, Src1, ShiftAmt));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
|
||||
}
|
||||
auto LastBit = _Lshr(CFSize, Src1, ShiftAmt);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
|
||||
|
||||
CalculatePF(Res, Src2);
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// OF
|
||||
{
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// OF flag is set if a sign change occurred
|
||||
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
|
||||
}
|
||||
|
||||
// Now select between the two
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
auto val = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
|
||||
// SF/ZF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
auto OldNZCV = GetNZCV();
|
||||
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
|
||||
// SF/ZF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto CFSize = IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1)));
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, One);
|
||||
auto LastBit = _Bfe(CFSize, 1, 0, _Lshr(CFSize, Src1, ShiftAmt));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
|
||||
}
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, _Constant(1));
|
||||
auto LastBit = _Lshr(CFSize, Src1, ShiftAmt);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
|
||||
|
||||
CalculatePF(Res, Src2);
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Now select between the two
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *UnmaskedRes, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetNZ_ZeroCV(SrcSize, UnmaskedRes);
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -805,10 +773,10 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
|
||||
if (SrcSizeBits < Shift) {
|
||||
Shift &= (SrcSizeBits - 1);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(OpSize, 1, SrcSizeBits - Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
CalculatePF(UnmaskedRes);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
@@ -817,9 +785,8 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
|
||||
// OF
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
if (Shift == 1) {
|
||||
auto Xor = _Xor(OpSize, Res, Src1);
|
||||
auto OF = _Bfe(OpSize, 1, SrcSize * 8 - 1, Xor);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OF);
|
||||
auto Xor = _Xor(OpSize, UnmaskedRes, Src1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, SrcSize * 8 - 1, true);
|
||||
} else {
|
||||
// Undefined, we choose to zero as part of SetNZ_ZeroCV
|
||||
}
|
||||
@@ -834,7 +801,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1))), 1, Shift-1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift-1, true);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -850,8 +817,6 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
// Stash OF before overwriting it
|
||||
auto OldOF = Shift != 1 ? GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC) : NULL;
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -859,7 +824,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(OpSize, 1, Shift-1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift-1, true);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -878,7 +843,6 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
CalculateFlags_ShiftRightImmediateCommon(SrcSize, Res, Src1, Shift);
|
||||
|
||||
// OF
|
||||
@@ -886,7 +850,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// Is set to the MSB of the original value
|
||||
if (Shift == 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Src1, SrcSize * 8 - 1, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -904,62 +868,53 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
// Is set if the MSB bit changes.
|
||||
// XOR of Result and Src1
|
||||
if (Shift == 1) {
|
||||
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
|
||||
auto val = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
|
||||
auto SizeBits = SrcSize * 8;
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
auto OldNZCV = GetNZCV();
|
||||
auto OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
ZeroCV();
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
auto NewCF = _Bfe(OpSize, 1, SizeBits - 1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
|
||||
// Now select: if shift == 0, don't update flags
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
|
||||
auto OldNZCV = GetNZCV();
|
||||
auto OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// Ends up faster overall.
|
||||
// XXX: can do much better if we have FlagM (with RMIF).
|
||||
ZeroCV();
|
||||
// Extract the last bit shifted in to CF
|
||||
//auto Size = _Constant(GetSrcSize(Res) * 8);
|
||||
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
//auto Size = _Constant(GetSrcSize(Res) * 8);
|
||||
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
|
||||
auto NewCF = _Bfe(OpSize, 1, 0, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 1, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
|
||||
// Now select: if shift == 0, don't update flags
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
@@ -967,16 +922,16 @@ void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, Ord
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
auto NewCF = _Bfe(OpSize, 1, SizeBits - 1, Res);
|
||||
|
||||
// Ends up faster overall. If Shift != 1, OF is undefined so we choose to zero here.
|
||||
// XXX: can do much better if we have FlagM (with RMIF).
|
||||
ZeroCV();
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
|
||||
}
|
||||
|
||||
// OF
|
||||
@@ -984,8 +939,8 @@ void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, Ord
|
||||
if (Shift == 1) {
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -996,16 +951,15 @@ void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, Orde
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(OpSize, 1, 0, Res);
|
||||
|
||||
// Ends up faster overall. If Shift != 1, OF is undefined so we choose to zero here.
|
||||
// XXX: can do much better if we have FlagM (with RMIF).
|
||||
ZeroCV();
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
|
||||
}
|
||||
|
||||
// OF
|
||||
@@ -1014,37 +968,13 @@ void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, Orde
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 1, Res), NewCF);
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
|
||||
// Zero AF. Note that we set the PF byte to 0/1 above, so PF[4] is 0 so the
|
||||
// XOR with PF will have no effect, so setting the AF byte to zero will indeed
|
||||
// zero AF as intended.
|
||||
uint32_t FlagsMaskToZero =
|
||||
(1U << X86State::RFLAG_AF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_SF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
ZeroMultipleFlags(FlagsMaskToZero);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode *Src) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
@@ -1155,34 +1085,12 @@ void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Src) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
// Now for the flags
|
||||
|
||||
auto Bounds = _Constant(SrcSize * 8- 1);
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
// OF cleared
|
||||
SetRFLAG<X86State::RFLAG_OF_RAW_LOC>(Zero);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Result, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_RAW_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_UGT,
|
||||
Src, Bounds,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_TZCNT(OrderedNode *Src) {
|
||||
@@ -1196,12 +1104,10 @@ void OpDispatchBuilder::CalculateFlags_TZCNT(OrderedNode *Src) {
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Bfe(OpSize::i32Bit, 1, 0, Src));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Src, 0, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src) {
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
// OF, SF, AF, PF all undefined
|
||||
ZeroNZCV();
|
||||
|
||||
@@ -1212,22 +1118,7 @@ void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src)
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BITSELECT(OrderedNode *Src) {
|
||||
// OF, SF, AF, PF, CF all undefined
|
||||
ZeroNZCV();
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
// ZF is set to 1 if the source was zero
|
||||
auto ZFSelectOp = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, ZeroConst,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(ZFSelectOp);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Src, SrcSize * 8 - 1, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
|
||||
|
||||
@@ -225,9 +225,7 @@ void OpDispatchBuilder::VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSi
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
auto ALUOp = _VAdd(Size, ElementSize, Dest, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Dest, Src));
|
||||
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
@@ -371,9 +369,7 @@ void OpDispatchBuilder::AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t Elemen
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
|
||||
auto ALUOp = _VAdd(Size, ElementSize, Src1, Src2);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Src1, Src2));
|
||||
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
@@ -506,9 +502,7 @@ void OpDispatchBuilder::VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementS
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
auto ALUOp = _VAdd(Size, ElementSize, Src, Dest);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Src, Dest));
|
||||
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
@@ -540,10 +534,8 @@ OrderedNode* OpDispatchBuilder::VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IR
|
||||
{.AllowUpperGarbage = true});
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
auto ALUOp = _VFAddScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
DeriveOp(ALUOp, IROp,
|
||||
_VFAddScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits));
|
||||
return ALUOp;
|
||||
}
|
||||
|
||||
@@ -626,10 +618,7 @@ OrderedNode* OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IRO
|
||||
{.AllowUpperGarbage = true});
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
auto ALUOp = _VFSqrtScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrtScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits));
|
||||
return ALUOp;
|
||||
}
|
||||
|
||||
@@ -940,9 +929,7 @@ void OpDispatchBuilder::VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t Element
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(OpSize, ElementSize, Src));
|
||||
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
@@ -979,9 +966,7 @@ void OpDispatchBuilder::AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t Elem
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(OpSize, ElementSize, Src));
|
||||
|
||||
// NOTE: We don't need to clear the upper lanes here, since the
|
||||
// IR ops make use of 128-bit AdvSimd for 128-bit cases,
|
||||
@@ -1017,9 +1002,7 @@ void OpDispatchBuilder::VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_
|
||||
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto ALUOp = _VFSqrt(ElementSize, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(ElementSize, ElementSize, Src));
|
||||
|
||||
// Duplicate the lower bits
|
||||
auto Result = _VDupElement(Size, ElementSize, ALUOp, 0);
|
||||
@@ -1746,8 +1729,7 @@ void OpDispatchBuilder::VHADDPOp(OpcodeArgs) {
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
|
||||
auto Res = _VFAddP(SrcSize, ElementSize, Src1, Src2);
|
||||
Res.first->Header.Op = IROp;
|
||||
DeriveOp(Res, IROp, _VFAddP(SrcSize, ElementSize, Src1, Src2));
|
||||
|
||||
OrderedNode *Dest = Res;
|
||||
if (Is256Bit) {
|
||||
@@ -2439,8 +2421,7 @@ void OpDispatchBuilder::AVXVariableShiftImpl(OpcodeArgs, IROps IROp) {
|
||||
OrderedNode *Vector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags);
|
||||
OrderedNode *ShiftVector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], DstSize, Op->Flags);
|
||||
|
||||
auto Shift = _VUShr(DstSize, SrcSize, Vector, ShiftVector, true);
|
||||
Shift.first->Header.Op = IROp;
|
||||
DeriveOp(Shift, IROp, _VUShr(DstSize, SrcSize, Vector, ShiftVector, true));
|
||||
|
||||
StoreResult(FPRClass, Op, Shift, -1);
|
||||
}
|
||||
@@ -3441,20 +3422,21 @@ void OpDispatchBuilder::VPALIGNROp(OpcodeArgs) {
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : GetSrcSize(Op);
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetGuestVectorLength(), Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
OrderedNode *Res = _FCmp(ElementSize, Src1, Src2,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
GenerateFlags_FCMP(Op, Res, Src1, Src2);
|
||||
CachedNZCV = nullptr;
|
||||
_FCmp(ElementSize, Src1, Src2);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToSSE();
|
||||
|
||||
flagsOp = SelectionFlag::FCMP;
|
||||
flagsOpDest = Src1;
|
||||
flagsOpSrc = Src2;
|
||||
flagsOpSize = GetSrcSize(Op);
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so PF[4] is
|
||||
// 0 so the XOR with PF will have no effect, so setting the AF byte to zero
|
||||
// will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
template
|
||||
|
||||
@@ -247,10 +247,13 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
|
||||
// Extract sign and make interger absolute
|
||||
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
|
||||
// We're about to clobber flags to grab the sign, so save NZCV.
|
||||
SaveNZCV();
|
||||
|
||||
auto absolute = _Abs(OpSize::i64Bit, data);
|
||||
// Extract sign and make interger absolute
|
||||
_SubNZCV(OpSize::i64Bit, data, zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType{COND_SLT}, _Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, data, CondClassType{COND_MI});
|
||||
|
||||
// left justify the absolute interger
|
||||
auto shift = _Sub(OpSize::i64Bit, _Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
@@ -856,9 +859,7 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80Round(a);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F80Round(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
@@ -889,9 +890,7 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80Add(a, st1);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F80Add(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
|
||||
@@ -601,21 +601,13 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
auto low = _Constant(0);
|
||||
OrderedNode *data = _VCastFromGPR(8, 8, low);
|
||||
|
||||
OrderedNode *Res = _FCmp(8, a, data,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
// Now we do our comparison.
|
||||
_FCmp(8, a, data);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
|
||||
//TODO: This should obey rounding mode
|
||||
@@ -681,36 +673,22 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode *Res = _FCmp(8, a, b,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToSSE();
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -767,9 +745,7 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64SIN(a);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F64SIN(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
@@ -799,9 +775,7 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64ATAN(a, st1);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F64ATAN(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM ||
|
||||
IROp == IR::OP_F64FPREM1) {
|
||||
|
||||
@@ -1,41 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <unistd.h>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
|
||||
// Let the host take first stab at handling the signal
|
||||
auto Thread = GetTLSThread();
|
||||
HostSignalHandler &Handler = HostHandlers[Signal];
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
for (auto &Handler : Handler.Handlers) {
|
||||
if (Handler(Thread, Signal, Info, UContext)) {
|
||||
// If the host handler handled the fault then we can continue now
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Handler.FrontendHandler &&
|
||||
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Now let the frontend handle the signal
|
||||
// It's clearly a guest signal and this ends up being an OS specific issue
|
||||
HandleGuestSignal(Thread, Signal, Info, UContext);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,84 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#ifndef NDEBUG
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::X86Tables::X86InstDebugInfo {
|
||||
void InstallDebugInfo() {
|
||||
const std::tuple<uint8_t, uint8_t, Flags> BaseOpTable[] = {
|
||||
{0x50, 8, {FLAGS_MEM_ACCESS}},
|
||||
{0x58, 8, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0x68, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0x6A, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xAA, 4, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xC8, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xCC, 2, {FLAGS_DEBUG}},
|
||||
|
||||
{0xD7, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xF1, 1, {FLAGS_DEBUG}},
|
||||
{0xF4, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::tuple<uint8_t, uint8_t, Flags> TwoByteOpTable[] = {
|
||||
{0x0B, 1, {FLAGS_DEBUG}},
|
||||
{0x19, 7, {FLAGS_DEBUG}},
|
||||
{0x28, 2, {FLAGS_MEM_ALIGN_16}},
|
||||
|
||||
{0x31, 1, {FLAGS_DEBUG}},
|
||||
|
||||
{0xA2, 1, {FLAGS_DEBUG}},
|
||||
{0xA3, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xAB, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xB3, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xBB, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xFF, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::tuple<uint8_t, uint8_t, Flags> PrimaryGroupOpTable[] = {
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 2, {FLAGS_DIVIDE}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 2, {FLAGS_DIVIDE}},
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
const std::tuple<uint16_t, uint8_t, Flags> SecondaryExtensionOpTable[] = {
|
||||
#define PF_NONE 0
|
||||
#define PF_F3 1
|
||||
#define PF_66 2
|
||||
#define PF_F2 3
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, {FLAGS_DEBUG}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, {FLAGS_DEBUG}},
|
||||
#undef PF_F3
|
||||
#undef PF_66
|
||||
#undef PF_F2
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
auto GenerateDebugTable = [](auto& FinalTable, auto& LocalTable) {
|
||||
for (auto Op : LocalTable) {
|
||||
auto OpNum = std::get<0>(Op);
|
||||
auto DebugInfo = std::get<2>(Op);
|
||||
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
|
||||
memcpy(&FinalTable[OpNum+i].DebugInfo, &DebugInfo, sizeof(X86InstDebugInfo::Flags));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
GenerateDebugTable(BaseOps, BaseOpTable);
|
||||
GenerateDebugTable(SecondBaseOps, TwoByteOpTable);
|
||||
GenerateDebugTable(PrimaryInstGroupOps, PrimaryGroupOpTable);
|
||||
|
||||
GenerateDebugTable(SecondInstGroupOps, SecondaryExtensionOpTable);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
@@ -44,11 +44,6 @@ void InitializeVEXTables();
|
||||
void InitializeXOPTables();
|
||||
void InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
uint64_t Total{};
|
||||
uint64_t NumInsts{};
|
||||
#endif
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
@@ -62,10 +57,6 @@ void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeVEXTables();
|
||||
InitializeXOPTables();
|
||||
InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
X86InstDebugInfo::InstallDebugInfo();
|
||||
#endif
|
||||
}
|
||||
|
||||
}
|
||||
@@ -100,10 +100,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
|
||||
// This should just throw a GP
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
@@ -147,19 +147,19 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1, nullptr}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4, nullptr}},
|
||||
@@ -169,7 +169,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3, nullptr}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
@@ -192,8 +192,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -183,41 +183,41 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 10
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
// GROUP 12
|
||||
{OPD(TYPE_GROUP_12, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -29,7 +29,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0C, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0D, 1, X86InstInfo{"", TYPE_GROUP_P, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0E, 1, X86InstInfo{"FEMMS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -44,7 +44,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x16, 1, X86InstInfo{"MOVLHPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x17, 1, X86InstInfo{"MOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"", TYPE_GROUP_16, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_DEBUG | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0x20, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x22, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -59,7 +59,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x2F, 1, X86InstInfo{"COMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"WRMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_DEBUG | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x32, 1, X86InstInfo{"RDMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x33, 1, X86InstInfo{"RDPMC", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -166,7 +166,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SETLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"SETNLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_DEBUG | FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -254,7 +254,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xFC, 1, X86InstInfo{"PADDB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFD, 1, X86InstInfo{"PADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
@@ -279,9 +279,15 @@ namespace InstFlags {
|
||||
using InstFlagType = uint64_t;
|
||||
|
||||
constexpr InstFlagType FLAGS_NONE = 0;
|
||||
constexpr InstFlagType FLAGS_DEBUG = (1ULL << 1);
|
||||
// The secondary Opcode Map uses prefix bytes to overlay more instruction
|
||||
// But some instructions need to ignore this overlay and consume these prefixes.
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 0);
|
||||
// Some instructions partially ignore overlay
|
||||
// Ignore OpSize (0x66) in this case
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 1);
|
||||
constexpr InstFlagType FLAGS_DEBUG_MEM_ACCESS = (1ULL << 2);
|
||||
constexpr InstFlagType FLAGS_SUPPORTS_REP = (1ULL << 3);
|
||||
// Only SEXT if the instruction is operating in 64bit operand size
|
||||
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 3);
|
||||
constexpr InstFlagType FLAGS_BLOCK_END = (1ULL << 4);
|
||||
constexpr InstFlagType FLAGS_SETS_RIP = (1ULL << 5);
|
||||
|
||||
@@ -331,27 +337,17 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
|
||||
constexpr InstFlagType FLAGS_SF_MOD_MEM_ONLY = (1ULL << 18);
|
||||
constexpr InstFlagType FLAGS_SF_MOD_REG_ONLY = (1ULL << 19);
|
||||
|
||||
// The secondary Opcode Map uses prefix bytes to overlay more instruction
|
||||
// But some instructions need to ignore this overlay and consume these prefixes.
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 20);
|
||||
// Some instructions partially ignore overlay
|
||||
// Ignore OpSize (0x66) in this case
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 21);
|
||||
|
||||
// x87
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 22);
|
||||
|
||||
// Only SEXT if the instruction is operating in 64bit operand size
|
||||
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 23);
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
|
||||
|
||||
// Whether or not the instruction has a VEX prefix for the first source operand
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
|
||||
// Whether or not the instruction has a VEX prefix for the second source operand
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -419,35 +415,12 @@ constexpr uint8_t OpToIndex(uint8_t Op) {
|
||||
using DecodedOp = DecodedInst const*;
|
||||
using OpDispatchPtr = void (IR::OpDispatchBuilder::*)(DecodedOp);
|
||||
|
||||
#ifndef NDEBUG
|
||||
namespace X86InstDebugInfo {
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_4 = (1 << 0);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_8 = (1 << 1);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_16 = (1 << 2);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_SIZE = (1 << 3); // If instruction size changes depending on prefixes
|
||||
constexpr uint64_t FLAGS_MEM_ACCESS = (1 << 4);
|
||||
constexpr uint64_t FLAGS_DEBUG = (1 << 5);
|
||||
constexpr uint64_t FLAGS_DIVIDE = (1 << 6);
|
||||
|
||||
|
||||
struct Flags {
|
||||
uint64_t DebugFlags;
|
||||
};
|
||||
void InstallDebugInfo();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
struct X86InstInfo {
|
||||
char const *Name;
|
||||
InstType Type;
|
||||
InstFlags::InstFlagType Flags; ///< Must be larger than InstFlags enum
|
||||
uint8_t MoreBytes;
|
||||
OpDispatchPtr OpcodeDispatcher;
|
||||
#ifndef NDEBUG
|
||||
X86InstDebugInfo::Flags DebugInfo;
|
||||
uint32_t NumUnitTestsGenerated;
|
||||
#endif
|
||||
|
||||
bool operator==(const X86InstInfo &b) const {
|
||||
if (strcmp(Name, b.Name) != 0 ||
|
||||
@@ -524,12 +497,6 @@ extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
// EVEX
|
||||
extern std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
|
||||
|
||||
|
||||
#ifndef NDEBUG
|
||||
extern uint64_t Total;
|
||||
extern uint64_t NumInsts;
|
||||
#endif
|
||||
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
@@ -548,11 +515,6 @@ static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<Op
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -570,11 +532,6 @@ static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoS
|
||||
}
|
||||
else {
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -602,11 +559,6 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct
|
||||
}
|
||||
}
|
||||
}
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -363,18 +363,9 @@ namespace FEXCore::IR {
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), std::move(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->DebugStore.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
else {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -67,8 +67,6 @@
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_ANDZ = 14 /* (a & b) == 0 */",
|
||||
"constexpr uint8_t COND_ANDNZ = 15 /* (a & b) != 0 */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
@@ -77,6 +75,8 @@
|
||||
"constexpr uint8_t COND_FU = 20 /* float unordred */",
|
||||
"constexpr uint8_t COND_FNU = 21 /* float not unordred */",
|
||||
|
||||
"constexpr uint8_t COND_AL = 32 /* always */",
|
||||
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRClass {0}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
@@ -264,7 +264,7 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "0"
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}": {
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}, i1:$FromNZCV{false}": {
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2",
|
||||
"EmitValidation": [
|
||||
@@ -447,6 +447,17 @@
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = LoadNZCV": {
|
||||
"Desc": ["Loads value of NZCV register"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
|
||||
"StoreNZCV GPR:$Value": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores value to NZCV register"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
|
||||
"GPR = LoadFlag u32:$Flag": {
|
||||
"Desc": ["Loads an x86-64 flag from the context object",
|
||||
"Specialized to allow flexible implementation of flag handling"
|
||||
@@ -854,9 +865,9 @@
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Integer negation",
|
||||
"Dest = -Src",
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{{COND_AL}}": {
|
||||
"Desc": ["Integer negation, with optional predication",
|
||||
"Dest = Cond ? -Src : Src",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
@@ -864,17 +875,6 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Abs OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Integer 2's complement absolute value",
|
||||
"Dest = std::abs(Src)",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Not OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Integer binary not",
|
||||
"op:",
|
||||
@@ -953,28 +953,48 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AddNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"AddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the sum of two GPRs"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"CarryInvert": {
|
||||
"Desc": ["Invert carry flag in NZCV"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"AXFlag": {
|
||||
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": {
|
||||
"Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"CondAddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
|
||||
"Desc": ["If condition is true, set NZCV per sum of GPRs, else force NZCV to a constant."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = SbbNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"AdcNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"SbbNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Sub OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -996,14 +1016,14 @@
|
||||
"_Shift != ShiftType::ROR"
|
||||
]
|
||||
},
|
||||
"GPR = SubNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, u8:$InvertCarry": {
|
||||
"Desc": ["Return NZCV for the difference of two GPRs. ",
|
||||
"If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.",
|
||||
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs. ",
|
||||
"Carry flag uses arm64 definition, inverted x86.",
|
||||
""],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Or OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -1046,6 +1066,13 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = XorShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
|
||||
"Desc": [ "Integer binary exclusive or with shifted register"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary and"
|
||||
],
|
||||
@@ -1061,10 +1088,10 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "4"
|
||||
"TestNZ OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the binary AND of two GPRs, setting N and Z accordingly and zeroing C and V"],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
@@ -1209,6 +1236,16 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = NZCVSelect OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Select based on value in NZCV flags",
|
||||
"op:",
|
||||
"Dest = Cond ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"EmitValidation": [
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Select OpSize:#ResultSize, OpSize:$CompareSize, CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
@@ -1320,12 +1357,12 @@
|
||||
"DestSize": "DestElementSize"
|
||||
},
|
||||
|
||||
"GPR = FCmp u8:$ElementSize, FPR:$Scalar1, FPR:$Scalar2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"FCmp u8:$ElementSize, FPR:$Scalar1, FPR:$Scalar2": {
|
||||
"Desc": ["Does a scalar unordered compare and sets NZCV accordingly.",
|
||||
"NZCV follows Arm conventions, a separate AXFLAG instruction is required for x86",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "4"
|
||||
"HasSideEffects": true
|
||||
}
|
||||
},
|
||||
"VectorScalar": {
|
||||
@@ -1581,7 +1618,6 @@
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VCMPEQZ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1590,7 +1626,6 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1599,7 +1634,6 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1616,6 +1650,10 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShraI u8:#RegisterSize, u8:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSShrI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1848,12 +1886,10 @@
|
||||
},
|
||||
|
||||
"FPR = VFMin u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMax u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1964,8 +2000,7 @@
|
||||
|
||||
"FPR = VCMPEQ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VCMPGT u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
@@ -1973,7 +2008,6 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
|
||||
],
|
||||
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -2153,6 +2187,14 @@
|
||||
"Desc": "Assists in key generation",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VSha1H FPR:$Src": {
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i32Bit"
|
||||
},
|
||||
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U0 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit"
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, u8:$SrcSize": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
|
||||
@@ -44,6 +44,11 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
|
||||
if (Arg == COND_AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
}
|
||||
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
|
||||
@@ -194,8 +194,6 @@ public:
|
||||
|
||||
private:
|
||||
bool HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
@@ -267,92 +265,6 @@ bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& Current
|
||||
return Changed;
|
||||
}
|
||||
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)
|
||||
&& !ImplicitFlagClobber(UnaryOpHdr->Op)) {
|
||||
// could be moved
|
||||
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
|
||||
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
|
||||
auto SelectOp = SelectOpHdr->CW<IR::IROp_Select>();
|
||||
|
||||
// the value isn't used after the select otherwise
|
||||
// make sure the sizes match
|
||||
if (SelectOpHdr->Size == UnaryOpHdr->Size && SelectOpHdr->Op == OP_SELECT && SelectOpNode->NumUses == 1
|
||||
&& IREmit->IsValueConstant(SelectOp->TrueVal)
|
||||
&& IREmit->IsValueConstant(SelectOp->FalseVal)) {
|
||||
|
||||
IREmit->SetWriteCursor(IREmit->UnwrapNode(SelectOpNode->Header.Previous));
|
||||
|
||||
size_t OpSize = FEXCore::IR::GetSize(UnaryOpHdr->Op);
|
||||
|
||||
/// copy for TrueVal ///
|
||||
auto NewUnaryOp1 = IREmit->AllocateRawOp(OpSize);
|
||||
|
||||
// Copy over the op
|
||||
memcpy(NewUnaryOp1.first, UnaryOpHdr, OpSize);
|
||||
|
||||
for (int i = 0; i < IR::GetArgs(NewUnaryOp1.first->Op); i++) {
|
||||
NewUnaryOp1.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
|
||||
}
|
||||
// Set New Op to operate on the constant
|
||||
IREmit->ReplaceNodeArgument(NewUnaryOp1, 0, IREmit->UnwrapNode(SelectOp->TrueVal));
|
||||
// Make select use the operated constant
|
||||
IREmit->ReplaceNodeArgument(SelectOpNode, 2, NewUnaryOp1);
|
||||
|
||||
/// copy for FalseVal ///
|
||||
auto NewUnaryOp2 = IREmit->AllocateRawOp(OpSize);
|
||||
|
||||
// Copy over the op
|
||||
memcpy(NewUnaryOp2.first, UnaryOpHdr, OpSize);
|
||||
|
||||
for (int i = 0; i < IR::GetArgs(NewUnaryOp2.first->Op); i++) {
|
||||
NewUnaryOp2.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
|
||||
}
|
||||
// Set New Op to operate on the constant
|
||||
IREmit->ReplaceNodeArgument(NewUnaryOp2, 0, IREmit->UnwrapNode(SelectOp->FalseVal));
|
||||
// Make select use the operated constant
|
||||
IREmit->ReplaceNodeArgument(SelectOpNode, 3, NewUnaryOp2);
|
||||
|
||||
// Replace uses of the defuct unary op w/ select
|
||||
IREmit->ReplaceAllUsesWithRange(UnaryOpNode, SelectOpNode, IREmit->GetIterator(IREmit->WrapNode(UnaryOpNode)), IREmit->GetIterator(BlockOp->Last));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Make all FCMPs set no flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_FCMP) {
|
||||
auto fcmp = IROp->CW<IR::IROp_FCmp>();
|
||||
fcmp->Flags = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Set needed flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_GETHOSTFLAG) {
|
||||
auto ghf = IROp->CW<IR::IROp_GetHostFlag>();
|
||||
|
||||
auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW<IR::IROp_FCmp>();
|
||||
LOGMAN_THROW_AA_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
|
||||
if(fcmp->Header.Op == OP_FCMP) {
|
||||
fcmp->Flags |= 1 << ghf->Flag;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// LoadMem / StoreMem imm pooling
|
||||
// If imms are close by, use address gen to generate the values instead of using a new imm
|
||||
void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
@@ -733,6 +645,8 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
/* TODO: restore this when we have rmif or something? */
|
||||
#if 0
|
||||
case OP_TESTNZ: {
|
||||
auto Op = IROp->CW<IR::IROp_TestNZ>();
|
||||
uint64_t Constant1{};
|
||||
@@ -747,6 +661,7 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
case OP_OR: {
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
uint64_t Constant1{};
|
||||
@@ -1005,37 +920,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
auto Select = IREmit->GetOpHeader(Op->Header.Args[0]);
|
||||
|
||||
uint64_t Constant;
|
||||
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
|
||||
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
|
||||
|
||||
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
|
||||
if (SelectCmpClass == GPRPairClass) {
|
||||
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
|
||||
break;
|
||||
}
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Select->Args[2], &Constant1) && IREmit->IsValueConstant(Select->Args[3], &Constant2)) {
|
||||
if (Constant1 == 1 && Constant2 == 0) {
|
||||
auto slc = Select->C<IR::IROp_Select>();
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(Select->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(Select->Args[1]));
|
||||
Op->Cond = slc->Cond;
|
||||
Op->CompareSize = slc->CompareSize;
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1102,16 +986,54 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (IsImmLogical(Constant1, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
|
||||
bool Bitwise = Op->Cond == COND_ANDZ ||
|
||||
Op->Cond == COND_ANDNZ;
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (Bitwise ? IsImmLogical(Constant1, IROp->Size * 8) : IsImmAddSub(Constant1)) {
|
||||
if (IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
@@ -1142,6 +1064,33 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
uint64_t Constant0{};
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1) &&
|
||||
Constant1 == 0)
|
||||
{
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant0) &&
|
||||
(Constant0 == 1 || Constant0 == AllOnes))
|
||||
{
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
@@ -1269,6 +1218,35 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1288,8 +1266,6 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
CodeMotionAroundSelects(IREmit, CurrentIR);
|
||||
FCMPOptimization(IREmit, CurrentIR);
|
||||
LoadMemStoreMemImmediatePooling(IREmit, CurrentIR);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
@@ -277,6 +277,24 @@ namespace {
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, pf_raw),
|
||||
sizeof(FEXCore::Core::CPUState::pf_raw),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, af_raw),
|
||||
sizeof(FEXCore::Core::CPUState::af_raw),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
@@ -419,6 +437,10 @@ namespace {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// PF/AF
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
@@ -518,12 +519,21 @@ namespace {
|
||||
const auto GetRegAndClassFromOffset = [&, this](uint32_t Offset) {
|
||||
const auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
|
||||
const auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
|
||||
const auto pf = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
const auto af = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
|
||||
const auto [beginFpr, endFpr] = GetFPRBeginAndEnd();
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr), "Unexpected Offset {}", Offset);
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr) || (Offset == pf) || (Offset == af), "Unexpected Offset {}", Offset);
|
||||
|
||||
if (Offset >= beginGpr && Offset < endGpr) {
|
||||
unsigned FlagOffset =
|
||||
Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount - 2;
|
||||
|
||||
if (Offset == pf) {
|
||||
return PhysicalRegister(GPRFixedClass, FlagOffset);
|
||||
} else if (Offset == af) {
|
||||
return PhysicalRegister(GPRFixedClass, FlagOffset + 1);
|
||||
} else if (Offset >= beginGpr && Offset < endGpr) {
|
||||
auto reg = (Offset - beginGpr) / Core::CPUState::GPR_REG_SIZE;
|
||||
return PhysicalRegister(GPRFixedClass, reg);
|
||||
} else if (Offset >= beginFpr && Offset < endFpr) {
|
||||
@@ -544,12 +554,21 @@ namespace {
|
||||
const auto GetStaticMapFromOffset = [&](uint32_t Offset) -> LiveRange** {
|
||||
const auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
|
||||
const auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
|
||||
const auto pf = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
const auto af = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
|
||||
const auto [beginFpr, endFpr] = GetFPRBeginAndEnd();
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr), "Unexpected Offset {}", Offset);
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr) || (Offset == pf) || (Offset == af), "Unexpected Offset {}", Offset);
|
||||
|
||||
if (Offset >= beginGpr && Offset < endGpr) {
|
||||
unsigned FlagOffset =
|
||||
Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount - 2;
|
||||
|
||||
if (Offset == pf) {
|
||||
return &StaticMaps[FlagOffset];
|
||||
} else if (Offset == af) {
|
||||
return &StaticMaps[FlagOffset + 1];
|
||||
} else if (Offset >= beginGpr && Offset < endGpr) {
|
||||
auto reg = (Offset - beginGpr) / Core::CPUState::GPR_REG_SIZE;
|
||||
return &StaticMaps[reg];
|
||||
} else if (Offset >= beginFpr && Offset < endFpr) {
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -272,7 +272,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -460,7 +460,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
|
||||
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -585,7 +585,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
@@ -155,13 +155,6 @@ namespace CPU {
|
||||
*/
|
||||
[[nodiscard]] virtual void *MapRegion(void *HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief This is post-setup initialization that is called just before code executino
|
||||
*
|
||||
* Guest memory is available at this point and ThreadState is valid
|
||||
*/
|
||||
virtual void Initialize() {}
|
||||
|
||||
/**
|
||||
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
|
||||
*
|
||||
|
||||
@@ -116,16 +116,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY static fextl::unique_ptr<FEXCore::Context::Context> CreateNewContext();
|
||||
|
||||
/**
|
||||
* @brief Post creation context initialization
|
||||
* Once configurations have been set, do the post-creation initialization with that configuration
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return true if we managed to initialize correctly
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool InitializeContext() = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows setting up in memory code and other things prior to launchign code execution
|
||||
*
|
||||
@@ -201,25 +191,6 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets the program exit status
|
||||
*
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return The program exit status
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual int GetProgramStatus() const = 0;
|
||||
|
||||
/**
|
||||
* @brief [[threadsafe]] Returns the ExitReason of the parent thread. Typically used for async result status
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return The ExitReason for the parentthread
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual ExitReason GetExitReason() = 0;
|
||||
|
||||
/**
|
||||
* @brief [[theadsafe]] Checks if the Context is either done working or paused(in the case of single stepping)
|
||||
*
|
||||
@@ -231,22 +202,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool IsDone() const = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets a copy the CPUState of the parent thread
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
* @param State The state object to populate
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void GetCPUState(FEXCore::Core::CPUState *State) const = 0;
|
||||
|
||||
/**
|
||||
* @brief Copies the CPUState provided to the parent thread
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
* @param State The satate object to copy from
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetCPUState(const FEXCore::Core::CPUState *State) = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to pass in a custom CPUBackend creation factory
|
||||
*
|
||||
@@ -270,12 +225,36 @@ namespace FEXCore::Context {
|
||||
///< State reconstruction helpers
|
||||
///< Reconstructs the guest RIP from the passed in thread context and related Host PC.
|
||||
FEX_DEFAULT_VISIBILITY virtual uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) = 0;
|
||||
///< Reconstructs a compacted EFLAGS from FEX's internal EFLAG representation.
|
||||
FEX_DEFAULT_VISIBILITY virtual uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
/**
|
||||
* @brief Reconstructs a compacted EFLAGS from FEX's internal EFLAG representation.
|
||||
*
|
||||
* @param Thread The thread getting the state reconstructed
|
||||
* @param WasInJIT If the code was in the JIT at the time.
|
||||
* @param HostGPRs The host Arm64 GPRs at the point of state inside the JIT.
|
||||
* @param PSTATE The Arm64 PState value.
|
||||
*
|
||||
* If WasInJIT is false then HostGPRs and PSTATE is ignored, with the assumption that the FEX JIT has already stored all state in to the
|
||||
* ThreadState object.
|
||||
*
|
||||
* @return x86 EFLAGS reconstructed
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) = 0;
|
||||
///< Sets FEX's internal EFLAGS representation to the passed in compacted form.
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) = 0;
|
||||
/**
|
||||
* @brief Create a new thread object that doesn't inherit any state.
|
||||
* Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The thread state to inherit from if not nullptr.
|
||||
* @param ParentTID The thread ID that the parent is inheriting from
|
||||
*
|
||||
* @return A new InternalThreadState object for using with a new guest thread.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState = nullptr, uint64_t ParentTID = 0) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InitializeThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
@@ -72,7 +72,7 @@ namespace FEXCore::Core {
|
||||
static_assert(std::is_trivially_copyable_v<NonAtomicRefCounter<uint64_t>>, "needs to be trivially copyable");
|
||||
static_assert(sizeof(NonAtomicRefCounter<uint64_t>) == sizeof(uint64_t), "Needs to be correct size");
|
||||
|
||||
struct FEX_PACKED CPUState {
|
||||
struct CPUState {
|
||||
// Allows more efficient handling of the register
|
||||
// file in the event AVX is not supported.
|
||||
union XMMRegs {
|
||||
@@ -102,6 +102,8 @@ namespace FEXCore::Core {
|
||||
uint64_t InlineJITBlockHeader{};
|
||||
XMMRegs xmm{};
|
||||
uint8_t flags[48]{};
|
||||
uint64_t pf_raw{};
|
||||
uint64_t af_raw{};
|
||||
uint64_t mm[8][2]{};
|
||||
|
||||
// 32bit x86 state
|
||||
@@ -335,7 +337,4 @@ namespace FEXCore::Core {
|
||||
static_assert(sizeof(CpuStateFrame::SynchronousFaultData) == 8, "This needs to be 8 bytes");
|
||||
static_assert(std::alignment_of_v<CpuStateFrame::SynchronousFaultDataStruct> == 8, "This needs to be 8 bytes");
|
||||
static_assert(offsetof(CpuStateFrame, SynchronousFaultData) % 8 == 0, "This needs to be aligned");
|
||||
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
|
||||
}
|
||||
@@ -35,8 +35,6 @@ namespace Core {
|
||||
#endif
|
||||
};
|
||||
}
|
||||
using HostSignalDelegatorFunction = std::function<bool(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext)>;
|
||||
|
||||
class SignalDelegator {
|
||||
public:
|
||||
virtual ~SignalDelegator() = default;
|
||||
@@ -49,16 +47,6 @@ namespace Core {
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleSignal(int Signal, void *Info, void *UContext);
|
||||
|
||||
/**
|
||||
* @brief Check to ensure the XID handler is still set to the FEX handler
|
||||
*
|
||||
@@ -67,12 +55,6 @@ namespace Core {
|
||||
*/
|
||||
virtual void CheckXIDHandler() = 0;
|
||||
|
||||
constexpr static size_t MAX_SIGNALS {64};
|
||||
|
||||
// Use the last signal just so we are less likely to ever conflict with something that the guest application is using
|
||||
// 64 is used internally by Valgrind
|
||||
constexpr static size_t SIGNAL_FOR_PAUSE {63};
|
||||
|
||||
struct SignalDelegatorConfig {
|
||||
bool StaticRegisterAllocation{};
|
||||
bool SupportsAVX{};
|
||||
@@ -124,31 +106,5 @@ namespace Core {
|
||||
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
|
||||
virtual FEXCore::Core::InternalThreadState *GetTLSThread() = 0;
|
||||
virtual void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
virtual void FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
|
||||
virtual void FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
|
||||
|
||||
private:
|
||||
struct HostSignalHandler {
|
||||
fextl::vector<FEXCore::HostSignalDelegatorFunction> Handlers{};
|
||||
FEXCore::HostSignalDelegatorFunction FrontendHandler{};
|
||||
};
|
||||
std::array<HostSignalHandler, MAX_SIGNALS + 1> HostHandlers{};
|
||||
|
||||
protected:
|
||||
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].Handlers.push_back(std::move(Func));
|
||||
}
|
||||
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].FrontendHandler = std::move(Func);
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -334,7 +334,7 @@ friend class FEXCore::IR::PassManager;
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
virtual void SaveNZCV() {
|
||||
virtual void SaveNZCV(IROps Op) {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
|
||||
@@ -1,338 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#ifndef _WIN32
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = delete;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::mutex Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = delete;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
void lock_shared() {
|
||||
Mutex.lock_shared();
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
Mutex.unlock_shared();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
return Mutex.try_lock();
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
return Mutex.try_lock_shared();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::shared_mutex Mutex;
|
||||
};
|
||||
#endif
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedDeferredSignalWithMutexBase(const ScopedDeferredSignalWithMutexBase&) = delete;
|
||||
ScopedDeferredSignalWithMutexBase& operator=(ScopedDeferredSignalWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedDeferredSignalWithMutexBase(ScopedDeferredSignalWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex}
|
||||
, Thread {rhs.Thread} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedDeferredSignalWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the recount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedDeferredSignalWithMutex = ScopedDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedDeferredSignalWithSharedLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithUniqueLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedDeferredSignalWithForkableMutex = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedDeferredSignalWithForkableSharedLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithForkableUniqueLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
|
||||
class ScopedSignalMasker final {
|
||||
public:
|
||||
ScopedSignalMasker() = default;
|
||||
|
||||
void Mask(uint64_t Mask) {
|
||||
#ifndef _WIN32
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker(ScopedSignalMasker &&rhs) = default;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker &&) = default;
|
||||
|
||||
void Unmask() {
|
||||
#ifndef _WIN32
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
private:
|
||||
#ifndef _WIN32
|
||||
uint64_t OriginalMask{};
|
||||
#endif
|
||||
};
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedPotentialDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
ScopedPotentialDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
if (Thread) {
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
else {
|
||||
Masker.Mask(Mask);
|
||||
}
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedPotentialDeferredSignalWithMutexBase(const ScopedPotentialDeferredSignalWithMutexBase&) = delete;
|
||||
ScopedPotentialDeferredSignalWithMutexBase& operator=(ScopedPotentialDeferredSignalWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedPotentialDeferredSignalWithMutexBase(ScopedPotentialDeferredSignalWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex}
|
||||
, Thread {rhs.Thread} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedPotentialDeferredSignalWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
// Unmask back to the original signal mask
|
||||
Masker.Unmask();
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
ScopedSignalMasker Masker;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedPotentialDeferredSignalWithMutex = ScopedPotentialDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithSharedLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedPotentialDeferredSignalWithForkableMutex = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithForkableSharedLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithForkableUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -0,0 +1,243 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <signal.h>
|
||||
#ifndef _WIN32
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
class ForkableUniqueMutex final : public std::mutex {
|
||||
public:
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final : public std::shared_mutex {
|
||||
public:
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
// Helper class to manage deferred signal refcounting within a block scope
|
||||
class DeferredSignalRefCountGuard final {
|
||||
public:
|
||||
explicit DeferredSignalRefCountGuard(FEXCore::Core::InternalThreadState *Thread) : Thread(Thread) {
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
DeferredSignalRefCountGuard(const DeferredSignalRefCountGuard&) = delete;
|
||||
DeferredSignalRefCountGuard& operator=(DeferredSignalRefCountGuard&) = delete;
|
||||
DeferredSignalRefCountGuard(DeferredSignalRefCountGuard&& rhs) : Thread(rhs.Thread) {
|
||||
rhs.Thread = nullptr;
|
||||
}
|
||||
|
||||
~DeferredSignalRefCountGuard() {
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
// Helper class to mask POSIX signals within a block scope
|
||||
class ScopedSignalMasker final {
|
||||
public:
|
||||
explicit ScopedSignalMasker(uint64_t Mask) : OriginalMask(0) {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &*OriginalMask, sizeof(*OriginalMask));
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker(ScopedSignalMasker&& rhs) : OriginalMask(rhs.OriginalMask) {
|
||||
rhs.OriginalMask.reset();
|
||||
}
|
||||
|
||||
~ScopedSignalMasker() {
|
||||
if (OriginalMask) {
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(*OriginalMask));
|
||||
}
|
||||
}
|
||||
private:
|
||||
std::optional<uint64_t> OriginalMask{};
|
||||
};
|
||||
#endif
|
||||
|
||||
/**
|
||||
* @brief Produces a wrapper object around a scoped lock of the given mutex
|
||||
* while ensuring POSIX signals are masked while the mutex is locked
|
||||
*
|
||||
* Use this to prevent reentrancy issues of C++ mutexes with certain signal handlers.
|
||||
* Common examples of such issues are:
|
||||
* - C++ mutexes not unlocking due to a signal handler calling longjmp from within a scope owning the mutex
|
||||
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
|
||||
* again before unlocking
|
||||
*
|
||||
* Ownership of the returned object may be moved, but it is NOT SAFE to move across threads.
|
||||
*/
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]] static auto MaskSignalsAndLockMutex(MutexType& mutex, uint64_t Mask = ~0ULL) {
|
||||
#ifndef _WIN32
|
||||
// Signals are masked first, and then the lock is acquired
|
||||
struct {
|
||||
ScopedSignalMasker mask;
|
||||
LockType<MutexType> lock;
|
||||
} scope_guard { ScopedSignalMasker { Mask }, LockType<MutexType> { mutex } };
|
||||
return scope_guard;
|
||||
#else
|
||||
// TODO: Doesn't block signals which may or may not cause issues.
|
||||
return LockType<MutexType> { mutex };
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Produces a wrapper object around a scoped lock of the given mutex
|
||||
* while bumping the Thread's deferred signal refcount while the mutex is
|
||||
* locked.
|
||||
*/
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]] static auto GuardSignalDeferringSection(MutexType& mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL) {
|
||||
// Refcount is incremented first, and then the lock is acquired.
|
||||
struct {
|
||||
std::optional<DeferredSignalRefCountGuard> refcount;
|
||||
LockType<MutexType> lock;
|
||||
} scope_guard = { DeferredSignalRefCountGuard { Thread }, LockType<MutexType> { mutex } };
|
||||
return scope_guard;
|
||||
}
|
||||
|
||||
// Like GuardSignalDeferringSection but falls back to masking signals when Thread is nullptr
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]] static auto GuardSignalDeferringSectionWithFallback(MutexType& mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL) {
|
||||
#ifndef _WIN32
|
||||
using ExtraGuard = std::variant<ScopedSignalMasker, DeferredSignalRefCountGuard>;
|
||||
#else
|
||||
using ExtraGuard = std::variant<std::monostate, DeferredSignalRefCountGuard>;
|
||||
#endif
|
||||
|
||||
struct {
|
||||
ExtraGuard refcount_or_mask;
|
||||
LockType<MutexType> lock;
|
||||
} scope_guard {
|
||||
Thread ? ExtraGuard { DeferredSignalRefCountGuard { Thread } }
|
||||
#ifndef _WIN32
|
||||
: ExtraGuard { ScopedSignalMasker { Mask } }
|
||||
#else
|
||||
: ExtraGuard { }
|
||||
#endif
|
||||
};
|
||||
scope_guard.lock = LockType<MutexType> { mutex };
|
||||
return scope_guard;
|
||||
}
|
||||
}
|
||||
@@ -1712,6 +1712,12 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Evaluate into flags") {
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Carry flag invert") {
|
||||
TEST_SINGLE(cfinv(), "cfinv");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Arm to eXternal FLAG") {
|
||||
TEST_SINGLE(axflag(), "axflag");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: eXternal to Arm FLAG") {
|
||||
TEST_SINGLE(xaflag(), "xaflag");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Conditional compare - register") {
|
||||
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::None, Condition::CC_AL), "ccmn w29, w28, #nzcv, al");
|
||||
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::Flag_N, Condition::CC_AL), "ccmn w29, w28, #Nzcv, al");
|
||||
|
||||
@@ -1,123 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#ifndef _WIN32
|
||||
#include <signal.h>
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FHU {
|
||||
/**
|
||||
* @brief A drop-in replacement for std::lock_guard that masks POSIX signals while the mutex is locked
|
||||
*
|
||||
* Use this class to prevent reentrancy issues of C++ mutexes with certain signal handlers.
|
||||
* Common examples of such issues are:
|
||||
* - C++ mutexes not unlocking due to a signal handler longjmping out of a scope owning the mutex
|
||||
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
|
||||
* again before unlocking
|
||||
*
|
||||
* Ownership of this object may be moved, but it is NOT SAFE to move across threads.
|
||||
*
|
||||
* Constructor order:
|
||||
* 1) Mask signals
|
||||
* 2) Lock Mutex
|
||||
*
|
||||
* Destructor Order:
|
||||
* 1) Unlock Mutex
|
||||
* 2) Unmask signals
|
||||
*/
|
||||
#ifndef _WIN32
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedSignalMaskWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex} {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
|
||||
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
|
||||
: OriginalMask {rhs.OriginalMask}, Mutex {rhs.Mutex} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedSignalMaskWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
}
|
||||
}
|
||||
private:
|
||||
uint64_t OriginalMask{};
|
||||
MutexType *Mutex;
|
||||
};
|
||||
#else
|
||||
// TODO: Doesn't block signals which may or may not cause issues.
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedSignalMaskWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, [[maybe_unused]] uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex} {
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
|
||||
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedSignalMaskWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
using ScopedSignalMaskWithMutex = ScopedSignalMaskWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedSignalMaskWithSharedLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithUniqueLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
using ScopedSignalMaskWithForkableMutex = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedSignalMaskWithForkableSharedLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithForkableUniqueLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -14,14 +14,12 @@ logger.setLevel(logging.ERROR)
|
||||
@dataclass
|
||||
class TestData:
|
||||
name: str
|
||||
optimal: int
|
||||
expectedinstructioncount: int
|
||||
code: bytes
|
||||
instructions: list
|
||||
def __init__(self, Name, Optimal, ExpectedInstructionCount, Code, Instructions):
|
||||
def __init__(self, Name, ExpectedInstructionCount, Code, Instructions):
|
||||
self.name = Name
|
||||
self.expectedinstructioncount = ExpectedInstructionCount
|
||||
self.optimal = Optimal
|
||||
self.code = Code
|
||||
self.instructions = Instructions
|
||||
|
||||
@@ -29,10 +27,6 @@ class TestData:
|
||||
def Name(self):
|
||||
return self.name
|
||||
|
||||
@property
|
||||
def Optimal(self):
|
||||
return self.optimal
|
||||
|
||||
@property
|
||||
def ExpectedInstructionCount(self):
|
||||
return self.expectedinstructioncount
|
||||
@@ -58,6 +52,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_RPRES = (1 << 7)
|
||||
FEATURE_FLAGM = (1 << 8)
|
||||
FEATURE_FLAGM2 = (1 << 9)
|
||||
FEATURE_CRYPTO = (1 << 10)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -70,6 +65,7 @@ HostFeaturesLookup = {
|
||||
"RPRES" : HostFeatures.FEATURE_RPRES,
|
||||
"FLAGM" : HostFeatures.FEATURE_FLAGM,
|
||||
"FLAGM2" : HostFeatures.FEATURE_FLAGM2,
|
||||
"CRYPTO" : HostFeatures.FEATURE_CRYPTO,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
@@ -112,15 +108,10 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
|
||||
for key, items in json_data["Instructions"].items():
|
||||
ExpectedInstructionCount = 0
|
||||
Optimal = 0
|
||||
Instructions = []
|
||||
if ("ExpectedInstructionCount" in items):
|
||||
ExpectedInstructionCount = int(items["ExpectedInstructionCount"])
|
||||
|
||||
if ("Optimal" in items):
|
||||
if items["Optimal"].upper() == "YES":
|
||||
Optimal = 1
|
||||
|
||||
if ("Skip" in items):
|
||||
if items["Skip"].upper() == "YES":
|
||||
continue
|
||||
@@ -163,7 +154,7 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
with open(tmp_asm_out, "rb") as tmp_asm_out_file:
|
||||
binary_hex = tmp_asm_out_file.read()
|
||||
|
||||
TestDataMap[TestName] = TestData(key, Optimal, ExpectedInstructionCount, binary_hex, Instructions)
|
||||
TestDataMap[TestName] = TestData(key, ExpectedInstructionCount, binary_hex, Instructions)
|
||||
|
||||
os.remove(tmp_asm)
|
||||
os.remove(tmp_asm_out)
|
||||
@@ -181,7 +172,6 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
# };
|
||||
# struct TestInfo {
|
||||
# char InstName[128];
|
||||
# uint64_t Optimal;
|
||||
# int64_t ExpectedInstructionCount;
|
||||
# uint64_t CodeSize;
|
||||
# uint64_t x86InstCount;
|
||||
@@ -208,7 +198,6 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
|
||||
# Add each test
|
||||
for key, item in TestDataMap.items():
|
||||
MemData += struct.pack('128s', item.Name.encode("ascii"))
|
||||
MemData += struct.pack('Q', item.Optimal)
|
||||
MemData += struct.pack('q', item.ExpectedInstructionCount)
|
||||
MemData += struct.pack('Q', len(item.Code))
|
||||
MemData += struct.pack('Q', len(item.Instructions))
|
||||
|
||||
@@ -104,6 +104,17 @@ namespace FEXServerClient {
|
||||
return GetServerLockFolder() + "RootFS.lock";
|
||||
}
|
||||
|
||||
fextl::string GetTempFolder() {
|
||||
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
|
||||
if (XDGRuntimeEnv) {
|
||||
// If the XDG runtime directory works then use that.
|
||||
return XDGRuntimeEnv;
|
||||
}
|
||||
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
|
||||
// Might not be ideal but we don't have much of a choice.
|
||||
return fextl::string{std::filesystem::temp_directory_path().string()};
|
||||
}
|
||||
|
||||
fextl::string GetServerMountFolder() {
|
||||
// We need a FEXServer mount directory that has some tricky requirements.
|
||||
// - We don't want to use `/tmp/` if possible.
|
||||
@@ -119,17 +130,7 @@ namespace FEXServerClient {
|
||||
// - If this path doesn't exist then fallback to `/tmp/` as a last resort.
|
||||
// - pressure-vessel explicitly creates an internal XDG_RUNTIME_DIR inside its chroot.
|
||||
// - This is okay since pressure-vessel rbinds the FEX rootfs from the host to `/run/pressure-vessel/interpreter-root`.
|
||||
fextl::string Folder{};
|
||||
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
|
||||
if (XDGRuntimeEnv) {
|
||||
// If the XDG runtime directory works then use that.
|
||||
Folder = XDGRuntimeEnv;
|
||||
}
|
||||
else {
|
||||
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
|
||||
// Might not be ideal but we don't have much of a choice.
|
||||
Folder = std::filesystem::temp_directory_path().string();
|
||||
}
|
||||
auto Folder = GetTempFolder();
|
||||
|
||||
if (FEXCore::Config::FindContainer() == "pressure-vessel") {
|
||||
// In pressure-vessel the mount point changes location.
|
||||
|
||||
@@ -50,6 +50,7 @@ namespace FEXServerClient {
|
||||
fextl::string GetServerLockFolder();
|
||||
fextl::string GetServerLockFile();
|
||||
fextl::string GetServerRootFSLockFile();
|
||||
fextl::string GetTempFolder();
|
||||
fextl::string GetServerMountFolder();
|
||||
fextl::string GetServerSocketName();
|
||||
int GetServerFD();
|
||||
|
||||
@@ -227,7 +227,6 @@ void AssertHandler(char const *Message) {
|
||||
|
||||
struct TestInfo {
|
||||
char TestInst[128];
|
||||
uint64_t Optimal;
|
||||
int64_t ExpectedInstructionCount;
|
||||
uint64_t CodeSize;
|
||||
uint64_t x86InstCount;
|
||||
@@ -276,8 +275,8 @@ static bool TestInstructions(FEXCore::Context::Context *CTX, FEXCore::Core::Inte
|
||||
|
||||
LogMan::Msg::IFmt("Testing instruction '{}': {} host instructions", CurrentTest->TestInst, INSTStats->first.HostCodeInstructions);
|
||||
|
||||
// Show the code if we know the implementation isn't optimal or if the count of instructions changed to something we didn't expect.
|
||||
bool ShouldShowCode = CurrentTest->Optimal == 0 ||
|
||||
// Show the code if the count of instructions changed to something we didn't expect.
|
||||
bool ShouldShowCode =
|
||||
INSTStats->first.HostCodeInstructions != CurrentTest->ExpectedInstructionCount;
|
||||
|
||||
if (ShouldShowCode) {
|
||||
@@ -467,6 +466,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEATURE_RPRES = (1U << 7),
|
||||
FEATURE_FLAGM = (1U << 8),
|
||||
FEATURE_FLAGM2 = (1U << 9),
|
||||
FEATURE_CRYPTO = (1U << 10),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -503,11 +503,12 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FLAGM2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFLAGM2);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_CRYPTO) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECRYPTO);
|
||||
}
|
||||
|
||||
// Always enable ARMv8.1 LSE atomics.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEATOMICS);
|
||||
// Always enable crypto extensions.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECRYPTO);
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_SVE128) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLESVE);
|
||||
@@ -539,6 +540,9 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FLAGM2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFLAGM2);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_CRYPTO) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLECRYPTO);
|
||||
}
|
||||
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_HOSTFEATURES, fextl::fmt::format("{}", HostFeatureControl));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_FORCESVEWIDTH, fextl::fmt::format("{}", SVEWidth));
|
||||
@@ -549,7 +553,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
// Create FEXCore context.
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
|
||||
CTX->InitializeContext();
|
||||
auto SignalDelegation = FEX::DummyHandlers::CreateSignalDelegator();
|
||||
auto SyscallHandler = FEX::DummyHandlers::CreateSyscallHandler();
|
||||
|
||||
|
||||
@@ -34,21 +34,11 @@ class DummySignalDelegator final : public FEXCore::SignalDelegator, public FEXCo
|
||||
}
|
||||
|
||||
protected:
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext) override {}
|
||||
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread() override;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override {}
|
||||
void FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override {}
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread();
|
||||
};
|
||||
|
||||
fextl::unique_ptr<FEXCore::HLE::SyscallHandler> CreateSyscallHandler();
|
||||
|
||||
@@ -106,8 +106,7 @@ void AOTGenSection(FEXCore::Context::Context *CTX, ELFCodeLoader::LoadedSection
|
||||
setpriority(PRIO_PROCESS, FHU::Syscalls::gettid(), 19);
|
||||
|
||||
// Setup thread - Each compilation thread uses its own backing FEX thread
|
||||
FEXCore::Core::CPUState state;
|
||||
auto Thread = CTX->CreateThread(&state, FHU::Syscalls::gettid());
|
||||
auto Thread = CTX->CreateThread(0, 0);
|
||||
fextl::set<uint64_t> ExternalBranchesLocal;
|
||||
CTX->ConfigureAOTGen(Thread, &ExternalBranchesLocal, SectionMaxAddress);
|
||||
|
||||
|
||||
@@ -143,6 +143,10 @@ static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
|
||||
return GetMContext(ucontext)->regs[id];
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmPState(void* ucontext) {
|
||||
return GetMContext(ucontext)->pstate;
|
||||
}
|
||||
|
||||
static inline uint64_t *GetArmGPRs(void* ucontext) {
|
||||
return reinterpret_cast<uint64_t*>(GetMContext(ucontext)->regs);
|
||||
}
|
||||
@@ -313,6 +317,10 @@ static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmPState(void* ucontext) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint64_t *GetArmGPRs(void* ucontext) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Common/Config.h"
|
||||
#include "ELFCodeLoader.h"
|
||||
#include "VDSO_Emulation.h"
|
||||
#include "LinuxSyscalls/GdbServer.h"
|
||||
#include "LinuxSyscalls/LinuxAllocator.h"
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
#include "LinuxSyscalls/Utils/Threads.h"
|
||||
@@ -443,7 +444,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Context::InitializeStaticTables(Loader.Is64BitMode() ? FEXCore::Context::MODE_64BIT : FEXCore::Context::MODE_32BIT);
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
CTX->InitializeContext();
|
||||
|
||||
// Setup TSO hardware emulation immediately after initializing the context.
|
||||
FEX::TSO::SetupTSOEmulation(CTX.get());
|
||||
@@ -474,7 +474,14 @@ int main(int argc, char **argv, char **const envp) {
|
||||
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
fextl::unique_ptr<FEX::GdbServer> DebugServer;
|
||||
if (GdbServer) {
|
||||
DebugServer = fextl::make_unique<FEX::GdbServer>(CTX.get(), SignalDelegation.get(), SyscallHandler.get());
|
||||
}
|
||||
|
||||
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
// Pass in our VDSO thunks
|
||||
CTX->AppendThunkDefinitions(FEX::VDSO::GetVDSOThunkDefinitions());
|
||||
@@ -550,8 +557,9 @@ int main(int argc, char **argv, char **const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
auto ProgramStatus = CTX->GetProgramStatus();
|
||||
auto ProgramStatus = ParentThread->StatusCode;
|
||||
|
||||
DebugServer.reset();
|
||||
SyscallHandler.reset();
|
||||
SignalDelegation.reset();
|
||||
|
||||
|
||||
@@ -160,7 +160,6 @@ int main(int argc, char **argv, char **const envp)
|
||||
|
||||
FEXCore::Context::InitializeStaticTables();
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
CTX->InitializeContext();
|
||||
|
||||
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), {});
|
||||
|
||||
@@ -179,7 +178,7 @@ int main(int argc, char **argv, char **const envp)
|
||||
|
||||
if (Loader.LoadIR(CTX.get()))
|
||||
{
|
||||
CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
auto ShutdownReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
|
||||
@@ -211,10 +210,7 @@ int main(int argc, char **argv, char **const envp)
|
||||
LogMan::Msg::DFmt("Reason we left VM: {}", FEXCore::ToUnderlying(ShutdownReason));
|
||||
|
||||
// Just re-use compare state. It also checks against the expected values in config.
|
||||
FEXCore::Core::CPUState State;
|
||||
CTX->GetCPUState(&State);
|
||||
|
||||
const bool Passed = Loader.CompareStates(&State, SupportsAVX);
|
||||
const bool Passed = Loader.CompareStates(&ParentThread->CurrentFrame->State, SupportsAVX);
|
||||
|
||||
LogMan::Msg::IFmt("Passed? {}\n", Passed ? "Yes" : "No");
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
add_compile_options(-fno-operator-names)
|
||||
|
||||
set (SRCS
|
||||
GdbServer.cpp
|
||||
EmulatedFiles/EmulatedFiles.cpp
|
||||
FileManagement.cpp
|
||||
LinuxAllocator.cpp
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -46,9 +47,22 @@ namespace FEX::EmulatedFile {
|
||||
*
|
||||
* @return A temporary file that we can use
|
||||
*/
|
||||
static int GenTmpFD() {
|
||||
int fd = open("/tmp", O_RDWR | O_TMPFILE | O_EXCL, S_IRUSR | S_IWUSR);
|
||||
return fd;
|
||||
static int GenTmpFD(const char *pathname, int flags) {
|
||||
uint32_t memfd_flags {MFD_ALLOW_SEALING};
|
||||
if (flags & O_CLOEXEC) memfd_flags |= MFD_CLOEXEC;
|
||||
|
||||
return memfd_create(pathname, memfd_flags);
|
||||
}
|
||||
|
||||
// Seal the tmpfd features by sealing them all.
|
||||
// Makes the tmpfd read-only.
|
||||
static void SealTmpFD(int fd) {
|
||||
fcntl(fd, F_ADD_SEALS,
|
||||
F_SEAL_SEAL |
|
||||
F_SEAL_SHRINK |
|
||||
F_SEAL_GROW |
|
||||
F_SEAL_WRITE |
|
||||
F_SEAL_FUTURE_WRITE);
|
||||
}
|
||||
|
||||
fextl::string GenerateCPUInfo(FEXCore::Context::Context *ctx, uint32_t CPUCores) {
|
||||
@@ -621,21 +635,23 @@ namespace FEX::EmulatedFile {
|
||||
}
|
||||
|
||||
EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx} {
|
||||
: CTX {ctx}
|
||||
, ThreadsConfig { FEXCore::CPUInfo::CalculateNumberOfCPUs() } {
|
||||
FDReadCreators["/proc/cpuinfo"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
// Only allow a single thread to initialize the cpu_info.
|
||||
// Jit in-case multiple threads try to initialize at once.
|
||||
// Check if deferred cpuinfo initialization has occured.
|
||||
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig()); });
|
||||
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig); });
|
||||
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
write(FD, (void*)&cpu_info.at(0), cpu_info.size());
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
FDReadCreators["/proc/sys/kernel/osrelease"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
uint32_t GuestVersion = FEX::HLE::_SyscallHandler->GetGuestKernelVersion();
|
||||
char Tmp[64]{};
|
||||
snprintf(Tmp, sizeof(Tmp), "%d.%d.%d\n",
|
||||
@@ -645,11 +661,12 @@ namespace FEX::EmulatedFile {
|
||||
// + 1 to ensure null at the end
|
||||
write(FD, Tmp, strlen(Tmp) + 1);
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
FDReadCreators["/proc/version"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
// UTS version NEEDS to be in a format that can pass to `date -d`
|
||||
// Format of this is Linux version <Release> (<Compile By>@<Compile Host>) (<Linux Compiler>) #<version> {SMP, PREEMPT, PREEMPT_RT} <UTS version>\n"
|
||||
const char kernel_version[] = "Linux version %d.%d.%d (FEX@FEX) (clang) #" GIT_DESCRIBE_STRING " SMP " __DATE__ " " __TIME__ "\n";
|
||||
@@ -662,13 +679,15 @@ namespace FEX::EmulatedFile {
|
||||
// + 1 to ensure null at the end
|
||||
write(FD, Tmp, strlen(Tmp) + 1);
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
auto NumCPUCores = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
write(FD, (void*)&cpus_online.at(0), cpus_online.size());
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
@@ -681,7 +700,7 @@ namespace FEX::EmulatedFile {
|
||||
FDReadCreators["/proc/self/auxv"] = &EmulatedFDManager::ProcAuxv;
|
||||
|
||||
auto cmdline_handler = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
auto CodeLoader = FEX::HLE::_SyscallHandler->GetCodeLoader();
|
||||
auto Args = CodeLoader->GetApplicationArguments();
|
||||
char NullChar{};
|
||||
@@ -695,6 +714,7 @@ namespace FEX::EmulatedFile {
|
||||
|
||||
// One additional null terminator to finish the list
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
@@ -702,9 +722,8 @@ namespace FEX::EmulatedFile {
|
||||
fextl::string procCmdLine = fextl::fmt::format("/proc/{}/cmdline", getpid());
|
||||
FDReadCreators[procCmdLine] = cmdline_handler;
|
||||
|
||||
uint64_t CPUCores = ThreadsConfig();
|
||||
if (CPUCores > 1) {
|
||||
cpus_online = fextl::fmt::format("0-{}", CPUCores - 1);
|
||||
if (ThreadsConfig > 1) {
|
||||
cpus_online = fextl::fmt::format("0-{}", ThreadsConfig - 1);
|
||||
}
|
||||
else {
|
||||
cpus_online = "0";
|
||||
@@ -721,6 +740,7 @@ namespace FEX::EmulatedFile {
|
||||
auto Creator = FDReadCreators.end();
|
||||
if (pathname) {
|
||||
Creator = FDReadCreators.find(pathname);
|
||||
Path = pathname;
|
||||
}
|
||||
|
||||
if (Creator == FDReadCreators.end()) {
|
||||
@@ -788,9 +808,10 @@ namespace FEX::EmulatedFile {
|
||||
return -1;
|
||||
}
|
||||
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
write(FD, (void*)auxvBase, auxvSize);
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -34,6 +34,6 @@ namespace FEX::EmulatedFile {
|
||||
fextl::unordered_map<fextl::string, FDReadStringFunc> FDReadCreators;
|
||||
|
||||
static int32_t ProcAuxv(FEXCore::Context::Context* ctx, int32_t fd, const char* pathname, int32_t flags, mode_t mode);
|
||||
FEX_CONFIG_OPT(ThreadsConfig, THREADS);
|
||||
const uint32_t ThreadsConfig;
|
||||
};
|
||||
}
|
||||
+217
-48
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
|
||||
#include <Common/FEXServerClient.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -32,6 +33,7 @@ $end_info$
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
@@ -42,18 +44,72 @@ $end_info$
|
||||
#endif
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <fmt/format.h>
|
||||
#include <poll.h>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <string_view>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/un.h>
|
||||
#include <sys/utsname.h>
|
||||
#include <unistd.h>
|
||||
#include <utility>
|
||||
|
||||
#include "GdbServer.h"
|
||||
#include "LinuxSyscalls/GdbServer.h"
|
||||
|
||||
namespace FEXCore
|
||||
namespace FEX
|
||||
{
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
"PF",
|
||||
"",
|
||||
"AF",
|
||||
"",
|
||||
"ZF",
|
||||
"SF",
|
||||
"TF",
|
||||
"IF",
|
||||
"DF",
|
||||
"OF",
|
||||
"IOPL",
|
||||
"",
|
||||
"NT",
|
||||
"",
|
||||
"RF",
|
||||
"VM",
|
||||
"AC",
|
||||
"VIF",
|
||||
"VIP",
|
||||
"ID",
|
||||
};
|
||||
|
||||
static std::string_view const& GetFlagName(unsigned Flag) {
|
||||
return FlagNames[Flag];
|
||||
}
|
||||
|
||||
static std::string_view const GetGRegName(unsigned Reg) {
|
||||
switch (Reg) {
|
||||
case FEXCore::X86State::REG_RAX: return "rax";
|
||||
case FEXCore::X86State::REG_RBX: return "rbx";
|
||||
case FEXCore::X86State::REG_RCX: return "rcx";
|
||||
case FEXCore::X86State::REG_RDX: return "rdx";
|
||||
case FEXCore::X86State::REG_RSP: return "rsp";
|
||||
case FEXCore::X86State::REG_RBP: return "rbp";
|
||||
case FEXCore::X86State::REG_RSI: return "rsi";
|
||||
case FEXCore::X86State::REG_RDI: return "rdi";
|
||||
case FEXCore::X86State::REG_R8: return "r8";
|
||||
case FEXCore::X86State::REG_R9: return "r9";
|
||||
case FEXCore::X86State::REG_R10: return "r10";
|
||||
case FEXCore::X86State::REG_R11: return "r11";
|
||||
case FEXCore::X86State::REG_R12: return "r12";
|
||||
case FEXCore::X86State::REG_R13: return "r13";
|
||||
case FEXCore::X86State::REG_R14: return "r14";
|
||||
case FEXCore::X86State::REG_R15: return "r15";
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
void GdbServer::Break(int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
@@ -70,7 +126,12 @@ void GdbServer::WaitForThreadWakeup() {
|
||||
ThreadBreakEvent.Wait();
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler)
|
||||
GdbServer::~GdbServer() {
|
||||
CoreShuttingDown = true;
|
||||
close(ListenSocket);
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler)
|
||||
: CTX(ctx)
|
||||
, SyscallHandler {SyscallHandler} {
|
||||
// Pass all signals by default
|
||||
@@ -88,7 +149,7 @@ GdbServer::GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDele
|
||||
|
||||
// This is a total hack as there is currently no way to resume once hitting a segfault
|
||||
// But it's semi-useful for debugging.
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
for (uint32_t Signal = 0; Signal <= FEX::HLE::SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, [this] (FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
if (PassSignals[Signal]) {
|
||||
// Pass signal to the guest
|
||||
@@ -140,6 +201,10 @@ static fextl::string encodeHex(const unsigned char *data, size_t length) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static fextl::string encodeHex(std::string_view str) {
|
||||
return encodeHex(reinterpret_cast<const unsigned char*>(str.data()), str.size());
|
||||
}
|
||||
|
||||
static fextl::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
fextl::string ThreadName;
|
||||
@@ -254,15 +319,15 @@ struct X80Float {
|
||||
};
|
||||
|
||||
struct FEX_PACKED GDBContextDefinition {
|
||||
uint64_t gregs[Core::CPUState::NUM_GPRS];
|
||||
uint64_t gregs[FEXCore::Core::CPUState::NUM_GPRS];
|
||||
uint64_t rip;
|
||||
uint32_t eflags;
|
||||
uint32_t cs, ss, ds, es, fs, gs;
|
||||
X80Float mm[Core::CPUState::NUM_MMS];
|
||||
X80Float mm[FEXCore::Core::CPUState::NUM_MMS];
|
||||
uint32_t fctrl;
|
||||
uint32_t fstat;
|
||||
uint32_t dummies[6];
|
||||
uint64_t xmm[Core::CPUState::NUM_XMMS][4];
|
||||
uint64_t xmm[FEXCore::Core::CPUState::NUM_XMMS][4];
|
||||
uint32_t mxcsr;
|
||||
};
|
||||
|
||||
@@ -293,9 +358,9 @@ fextl::string GdbServer::readRegs() {
|
||||
memcpy(&GDB.gregs[0], &state.gregs[0], sizeof(GDB.gregs));
|
||||
memcpy(&GDB.rip, &state.rip, sizeof(GDB.rip));
|
||||
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread);
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread, false, nullptr, 0);
|
||||
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
memcpy(&GDB.mm[i], &state.mm[i], sizeof(GDB.mm));
|
||||
}
|
||||
|
||||
@@ -349,7 +414,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
return {encodeHex((unsigned char *)(&state.rip), sizeof(uint64_t)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, eflags)) {
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread);
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread, false, nullptr, 0);
|
||||
|
||||
return {encodeHex((unsigned char *)(&eflags), sizeof(uint32_t)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -383,9 +448,9 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
}
|
||||
else if (addr >= offsetof(GDBContextDefinition, xmm[0][0]) &&
|
||||
addr < offsetof(GDBContextDefinition, xmm[16][0])) {
|
||||
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / FEXCore::Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto *Data = (unsigned char *)&state.xmm.avx.data[XmmIndex][0];
|
||||
return {encodeHex(Data, Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
|
||||
return {encodeHex(Data, FEXCore::Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, mxcsr)) {
|
||||
uint32_t Empty{};
|
||||
@@ -409,7 +474,7 @@ fextl::string buildTargetXML() {
|
||||
xml << "<flags id='fex_eflags' size='4'>\n";
|
||||
// flags register
|
||||
for(int i = 0; i < 22; i++) {
|
||||
auto name = FEXCore::Core::GetFlagName(i);
|
||||
auto name = GetFlagName(i);
|
||||
if (name.empty()) {
|
||||
continue;
|
||||
}
|
||||
@@ -427,8 +492,8 @@ fextl::string buildTargetXML() {
|
||||
// We want to just memcpy our x86 state to gdb, so we tell it the ordering.
|
||||
|
||||
// GPRs
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_GPRS; i++) {
|
||||
reg(FEXCore::Core::GetGRegName(i), "int64", 64);
|
||||
for (uint32_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; i++) {
|
||||
reg(GetGRegName(i), "int64", 64);
|
||||
}
|
||||
|
||||
reg("rip", "code_ptr", 64);
|
||||
@@ -484,7 +549,7 @@ fextl::string buildTargetXML() {
|
||||
)";
|
||||
|
||||
// SSE regs
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fextl::fmt::format("xmm{}", i), "vec128", 128);
|
||||
}
|
||||
|
||||
@@ -510,7 +575,7 @@ fextl::string buildTargetXML() {
|
||||
<field name="uint128" type="uint128"/>
|
||||
</union>
|
||||
)";
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fmt::format("ymm{}h", i), "vec128", 128);
|
||||
}
|
||||
xml << "</feature>\n";
|
||||
@@ -815,7 +880,6 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const fextl::string &packet
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const fextl::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
@@ -867,12 +931,17 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
SupportedFeatures += "QNonStop+;";
|
||||
|
||||
SupportedFeatures += "qXfer:osdata:read+;";
|
||||
SupportedFeatures += "QStartNoAckMode+;";
|
||||
|
||||
// Causes GDB to crash?
|
||||
// SupportedFeatures += "QStartNoAckMode+;";
|
||||
// TODO: Support breakpoints
|
||||
// SupportedFeatures += "swbreak+;";
|
||||
// SupportedFeatures += "hwbreak+;";
|
||||
// SupportedFeatures += "BreakpointCommands+;";
|
||||
|
||||
// TODO: If we want to support conditional breakpoints then we need to support single stepping.
|
||||
// SupportedFeatures += "ConditionalBreakpoints+;";
|
||||
|
||||
for (auto &Feature : Features) {
|
||||
|
||||
if (MatchStr(Feature, "swbreak+")) {
|
||||
SupportedFeatures += "swbreak+;";
|
||||
}
|
||||
@@ -972,13 +1041,68 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
// We now have a semi-colon deliminated list of signals to pass to the guest process
|
||||
for (fextl::string tmp; std::getline(ss, tmp, ';'); ) {
|
||||
uint32_t Signal = std::stoi(tmp.c_str(), nullptr, 16);
|
||||
if (Signal < SignalDelegator::MAX_SIGNALS) {
|
||||
if (Signal < FEX::HLE::SignalDelegator::MAX_SIGNALS) {
|
||||
PassSignals[Signal] = true;
|
||||
}
|
||||
}
|
||||
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
// lldb specific queries
|
||||
if (match("qHostInfo")) {
|
||||
// Returns Key:Value pairs separated by ;
|
||||
// eg:
|
||||
// triple:7838365f36342d70632d6c696e75782d676e75;
|
||||
// ptrsize:8;
|
||||
// distribution_id:7562756e7475;
|
||||
// watchpoint_exceptions_received:after;
|
||||
// endian:little;
|
||||
// os_version:6.3.3;
|
||||
// os_build:362e332e332d3036303330332d67656e65726963;
|
||||
// os_kernel:2332303233303531373133333620534d5020505245454d50545f44594e414d494320576564204d61792031372031333a34353a3139205554432032303233;
|
||||
// hostname:7279616e682d545235303030;
|
||||
fextl::string HostFeatures{};
|
||||
|
||||
// 64-bit always returned for the host environment.
|
||||
// qProcessInfo will return i386 or not.
|
||||
HostFeatures += fextl::fmt::format("triple:{};", encodeHex("x86_64-pc-linux-gnu"));
|
||||
HostFeatures += "ptrsize:8;";
|
||||
|
||||
// Always little-endian.
|
||||
HostFeatures += "endian:little;";
|
||||
|
||||
struct utsname buf{};
|
||||
if (uname(&buf) != -1) {
|
||||
uint32_t Major{};
|
||||
uint32_t Minor{};
|
||||
uint32_t Patch{};
|
||||
|
||||
// Parse kernel version in the form of `<Major>.<Minor>.<Patch>[Optional Data]`
|
||||
const auto End = buf.release + sizeof(buf.release);
|
||||
auto Results = std::from_chars(buf.release, End, Major, 10);
|
||||
Results = std::from_chars(Results.ptr + 1, End, Minor, 10);
|
||||
Results = std::from_chars(Results.ptr + 1, End, Patch, 10);
|
||||
|
||||
HostFeatures += fextl::fmt::format("os_version:{}.{}.{};", Major, Minor, Patch);
|
||||
|
||||
// os_build returns the release untouched.
|
||||
HostFeatures += fextl::fmt::format("os_build:{};", encodeHex(buf.release));
|
||||
HostFeatures += fextl::fmt::format("os_kernel:{};", encodeHex(buf.version));
|
||||
HostFeatures += fextl::fmt::format("hostname:{};", encodeHex(buf.nodename));
|
||||
}
|
||||
|
||||
// TODO: distribution_id should be fetched with `lsb_release -i`
|
||||
// TODO: watchpoint_exceptions_received is unsupported
|
||||
return {std::move(HostFeatures), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qGetWorkingDir")) {
|
||||
char Tmp[PATH_MAX];
|
||||
if (getcwd(Tmp, PATH_MAX)) {
|
||||
return {encodeHex(Tmp), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
@@ -1221,11 +1345,45 @@ void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
}
|
||||
}
|
||||
|
||||
GdbServer::WaitForConnectionResult GdbServer::WaitForConnection() {
|
||||
while (!CoreShuttingDown.load()) {
|
||||
struct pollfd PollFD {
|
||||
.fd = ListenSocket,
|
||||
.events = POLLIN | POLLPRI | POLLRDHUP,
|
||||
.revents = 0,
|
||||
};
|
||||
int Result = ppoll(&PollFD, 1, nullptr, nullptr);
|
||||
if (Result > 0) {
|
||||
if (PollFD.revents & POLLIN) {
|
||||
CommsStream = OpenSocket();
|
||||
return WaitForConnectionResult::CONNECTION;
|
||||
}
|
||||
else if (PollFD.revents & (POLLHUP | POLLERR | POLLNVAL)) {
|
||||
// Listen socket error or shutting down
|
||||
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}");
|
||||
return WaitForConnectionResult::ERROR;
|
||||
}
|
||||
}
|
||||
else if (Result == -1) {
|
||||
LogMan::Msg::EFmt("[GdbServer] poll failure: {}", errno);
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::EFmt("[GdbServer] Shutting Down");
|
||||
return WaitForConnectionResult::ERROR;
|
||||
}
|
||||
|
||||
void GdbServer::GdbServerLoop() {
|
||||
OpenListenSocket();
|
||||
if (ListenSocket == -1) {
|
||||
// Couldn't open socket, just exit.
|
||||
return;
|
||||
}
|
||||
|
||||
while (!CoreShuttingDown.load()) {
|
||||
CommsStream = OpenSocket();
|
||||
if (WaitForConnection() == WaitForConnectionResult::ERROR) {
|
||||
break;
|
||||
}
|
||||
|
||||
HandledPacketType response{};
|
||||
|
||||
@@ -1275,9 +1433,11 @@ void GdbServer::GdbServerLoop() {
|
||||
}
|
||||
|
||||
close(ListenSocket);
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::GdbServer *This = reinterpret_cast<FEXCore::GdbServer*>(Arg);
|
||||
FEXCore::Threads::SetThreadName("FEX:gdbserver");
|
||||
auto This = reinterpret_cast<FEX::GdbServer*>(Arg);
|
||||
This->GdbServerLoop();
|
||||
return nullptr;
|
||||
}
|
||||
@@ -1289,38 +1449,48 @@ void GdbServer::StartThread() {
|
||||
}
|
||||
|
||||
void GdbServer::OpenListenSocket() {
|
||||
// getaddrinfo allocates memory that can't be removed.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
struct addrinfo hints, *res;
|
||||
|
||||
memset(&hints, 0, sizeof(hints));
|
||||
hints.ai_family = AF_UNSPEC;
|
||||
hints.ai_socktype = SOCK_STREAM;
|
||||
hints.ai_flags = AI_PASSIVE;
|
||||
|
||||
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
|
||||
perror("getaddrinfo");
|
||||
const auto GdbUnixPath = fextl::fmt::format("{}/FEX_gdbserver/", FEXServerClient::GetTempFolder());
|
||||
if (FHU::Filesystem::CreateDirectory(GdbUnixPath) == FHU::Filesystem::CreateDirectoryResult::ERROR) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't create gdbserver folder {}", GdbUnixPath);
|
||||
return;
|
||||
}
|
||||
|
||||
int on = 1;
|
||||
GdbUnixSocketPath = fextl::fmt::format("{}{}-gdb", GdbUnixPath, ::getpid());
|
||||
|
||||
ListenSocket = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
|
||||
if (ListenSocket < 0) {
|
||||
perror("socket");
|
||||
ListenSocket = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (ListenSocket == -1) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
return;
|
||||
}
|
||||
if(setsockopt(ListenSocket, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
|
||||
perror("setsockopt");
|
||||
close(ListenSocket);
|
||||
|
||||
struct sockaddr_un addr{};
|
||||
addr.sun_family = AF_UNIX;
|
||||
strncpy(addr.sun_path, GdbUnixSocketPath.data(), sizeof(addr.sun_path));
|
||||
size_t SizeOfAddr = offsetof(sockaddr_un, sun_path) + GdbUnixSocketPath.size();
|
||||
|
||||
// Bind the socket to the path
|
||||
int Result{};
|
||||
for (int attempt = 0; attempt < 2; ++attempt) {
|
||||
Result = bind(ListenSocket, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
|
||||
if (Result == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
// This can happen periodically with execve. unlink the path and try again.
|
||||
// The PID is reused but FEX likely started a gdbserver thread for the PID before execve.
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
|
||||
if (bind(ListenSocket, res->ai_addr, res->ai_addrlen) < 0) {
|
||||
perror("bind");
|
||||
if (Result != 0) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't bind AF_UNIX socket '{}': {} {}\n", addr.sun_path, errno, strerror(errno));
|
||||
close(ListenSocket);
|
||||
ListenSocket = -1;
|
||||
return;
|
||||
}
|
||||
|
||||
listen(ListenSocket, 1);
|
||||
|
||||
freeaddrinfo(res);
|
||||
LogMan::Msg::IFmt("[GdbServer] Waiting for connection on {}", GdbUnixSocketPath);
|
||||
LogMan::Msg::IFmt("[GdbServer] gdb-multiarch -ex \"target extended-remote {}\"", GdbUnixSocketPath);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
@@ -1328,7 +1498,6 @@ fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
struct sockaddr_storage their_addr{};
|
||||
socklen_t addr_size{};
|
||||
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
return fextl::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
+12
-3
@@ -19,11 +19,14 @@ $end_info$
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
namespace FEX {
|
||||
|
||||
class GdbServer {
|
||||
public:
|
||||
GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
|
||||
GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
|
||||
~GdbServer();
|
||||
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
@@ -36,6 +39,11 @@ private:
|
||||
void Break(int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
enum class WaitForConnectionResult {
|
||||
CONNECTION,
|
||||
ERROR,
|
||||
};
|
||||
WaitForConnectionResult WaitForConnection();
|
||||
fextl::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
fextl::string ReadPacket(std::iostream &stream);
|
||||
@@ -90,9 +98,10 @@ private:
|
||||
fextl::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
std::array<bool, FEX::HLE::SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
fextl::string GdbUnixSocketPath{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
};
|
||||
@@ -150,6 +150,38 @@ namespace FEX::HLE {
|
||||
return SigInfoLayout::LAYOUT_KILL;
|
||||
}
|
||||
|
||||
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
|
||||
// Let the host take first stab at handling the signal
|
||||
auto Thread = GetTLSThread();
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
SignalHandler &Handler = HostHandlers[Signal];
|
||||
for (auto &HandlerFunc : Handler.Handlers) {
|
||||
if (HandlerFunc(Thread, Signal, Info, UContext)) {
|
||||
// If the host handler handled the fault then we can continue now
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Handler.FrontendHandler &&
|
||||
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Now let the frontend handle the signal
|
||||
// It's clearly a guest signal and this ends up being an OS specific issue
|
||||
HandleGuestSignal(Thread, Signal, Info, UContext);
|
||||
}
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
#ifdef _M_ARM_64
|
||||
for (size_t i = 0; i < Config.SRAGPRCount; i++) {
|
||||
@@ -1137,12 +1169,13 @@ namespace FEX::HLE {
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
const bool WasInJIT = Thread->CPUBackend->IsAddressInCodeBuffer(OldPC);
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (Config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
if (WasInJIT) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
@@ -1207,7 +1240,7 @@ namespace FEX::HLE {
|
||||
// Backup where we think the RIP currently is
|
||||
ContextBackup->OriginalRIP = CTX->RestoreRIPFromHostPC(Thread, ArchHelpers::Context::GetPc(ucontext));
|
||||
// Calculate eflags upfront.
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(Thread);
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(Thread, WasInJIT, ArchHelpers::Context::GetArmGPRs(ucontext), ArchHelpers::Context::GetArmPState(ucontext));
|
||||
|
||||
if (Is64BitMode) {
|
||||
NewGuestSP = SetupFrame_x64(Thread, ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
|
||||
@@ -1810,7 +1843,7 @@ namespace FEX::HLE {
|
||||
ThreadData.Thread = nullptr;
|
||||
}
|
||||
|
||||
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) {
|
||||
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
// Linux signal handlers are per-process rather than per thread
|
||||
// Multiple threads could be calling in to this
|
||||
std::lock_guard lk(HostDelegatorMutex);
|
||||
@@ -1818,7 +1851,7 @@ namespace FEX::HLE {
|
||||
InstallHostThunk(Signal);
|
||||
}
|
||||
|
||||
void SignalDelegator::FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) {
|
||||
void SignalDelegator::FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
// Linux signal handlers are per-process rather than per thread
|
||||
// Multiple threads could be calling in to this
|
||||
std::lock_guard lk(HostDelegatorMutex);
|
||||
|
||||
@@ -40,20 +40,36 @@ namespace FEX::HLE {
|
||||
|
||||
class SignalDelegator final : public FEXCore::SignalDelegator, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
constexpr static size_t MAX_SIGNALS {64};
|
||||
|
||||
// Use the last signal just so we are less likely to ever conflict with something that the guest application is using
|
||||
// 64 is used internally by Valgrind
|
||||
constexpr static size_t SIGNAL_FOR_PAUSE {63};
|
||||
|
||||
// Returns true if the host handled the signal
|
||||
// Arguments are the same as sigaction handler
|
||||
SignalDelegator(FEXCore::Context::Context *_CTX, const std::string_view ApplicationName);
|
||||
~SignalDelegator() override;
|
||||
|
||||
// Called from the signal trampoline function.
|
||||
void HandleSignal(int Signal, void *Info, void *UContext);
|
||||
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal specifically for guest handling
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void RegisterHostSignalHandlerForGuest(int Signal, FEX::HLE::HostSignalDelegatorFunctionForGuest Func);
|
||||
void RegisterHostSignalHandlerForGuest(int Signal, HostSignalDelegatorFunctionForGuest Func);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
/**
|
||||
@@ -107,21 +123,27 @@ namespace FEX::HLE {
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
|
||||
void SaveTelemetry();
|
||||
protected:
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext) override;
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread();
|
||||
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread() override;
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext);
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override;
|
||||
void FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override;
|
||||
void FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].Handlers.push_back(std::move(Func));
|
||||
}
|
||||
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].FrontendHandler = std::move(Func);
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
fextl::string const ApplicationName;
|
||||
@@ -155,6 +177,10 @@ namespace FEX::HLE {
|
||||
FEX::HLE::HostSignalDelegatorFunctionForGuest GuestHandler{};
|
||||
GuestSigAction GuestAction{};
|
||||
DefaultBehaviour DefaultBehaviour {DEFAULT_TERM};
|
||||
|
||||
// Callbacks
|
||||
fextl::vector<HostSignalDelegatorFunction> Handlers{};
|
||||
HostSignalDelegatorFunction FrontendHandler{};
|
||||
};
|
||||
|
||||
std::array<SignalHandler, MAX_SIGNALS + 1> HostHandlers{};
|
||||
|
||||
@@ -525,6 +525,14 @@ static uint64_t Clone3Handler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clo
|
||||
uint64_t CloneHandler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clone3_args *args) {
|
||||
uint64_t flags = args->args.flags;
|
||||
|
||||
if (flags & CLONE_CLEAR_SIGHAND) {
|
||||
// CLONE_CLEAR_SIGHAND was added in kernel 5.5. FEX doesn't properly support this.
|
||||
// glibc started using this flag in 2.38 as an optimization for posix_spawn.
|
||||
// If clone returns EINVAL or ENOSYS then it will fallback to the non-optimized path.
|
||||
LogMan::Msg::IFmt("CLONE_CLEAR_SIGHAND passed to clone3. Returning EINVAL.");
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
auto HasUnhandledFlags = [](FEX::HLE::clone3_args *args) -> bool {
|
||||
constexpr uint64_t UNHANDLED_FLAGS =
|
||||
CLONE_NEWNS |
|
||||
|
||||
@@ -16,7 +16,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -175,7 +175,6 @@ public:
|
||||
FEX_CONFIG_OPT(IsInterpreterInstalled, INTERPRETER_INSTALLED);
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThreadsConfig, THREADS);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
|
||||
|
||||
@@ -89,21 +89,8 @@ namespace FEX::HLE {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(getcpu, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, unsigned *cpu, unsigned *node, struct getcpu_cache *tcache) -> uint64_t {
|
||||
uint32_t LocalCPU{};
|
||||
uint32_t LocalNode{};
|
||||
// tcache is ignored
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(getcpu), cpu ? &LocalCPU : nullptr, node ? &LocalNode : nullptr, nullptr);
|
||||
if (Result == 0) {
|
||||
if (cpu) {
|
||||
// Ensure we don't return a number over our number of emulated cores
|
||||
*cpu = LocalCPU % FEX::HLE::_SyscallHandler->ThreadsConfig();
|
||||
}
|
||||
|
||||
if (node) {
|
||||
// Just claim we are part of node zero
|
||||
*node = 0;
|
||||
}
|
||||
}
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(getcpu), cpu, node, nullptr);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -80,35 +80,14 @@ namespace FEX::HLE {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(sched_setaffinity, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY | SyscallFlags::NOSIDEEFFECTS,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, pid_t pid, size_t cpusetsize, const unsigned long *mask) -> uint64_t {
|
||||
return 0;
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(sched_setaffinity), pid, cpusetsize, mask);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(sched_getaffinity, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, pid_t pid, size_t cpusetsize, unsigned char *mask) -> uint64_t {
|
||||
uint64_t Cores = FEX::HLE::_SyscallHandler->ThreadsConfig();
|
||||
|
||||
// Bytes need to round up to size of uint64_t
|
||||
uint64_t Bytes = FEXCore::AlignUp(Cores, sizeof(uint64_t));
|
||||
|
||||
// cpusetsize needs to be 8byte aligned
|
||||
if (cpusetsize & (sizeof(uint64_t) - 1)) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
// If we don't have enough bytes to store the resulting structure
|
||||
// then we need to return -EINVAL
|
||||
if (cpusetsize < Bytes) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
memset(mask, 0, Bytes);
|
||||
|
||||
for (uint64_t i = 0; i < Cores; ++i) {
|
||||
mask[i / 8] |= (1 << (i % 8));
|
||||
}
|
||||
|
||||
// Returns the number of bytes written in to mask
|
||||
return Bytes;
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(sched_getaffinity), pid, cpusetsize, mask);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(sched_setattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
|
||||
@@ -58,7 +58,7 @@ namespace FEX::HLE {
|
||||
NewThreadState.gregs[FEXCore::X86State::REG_RSP] = args->args.stack;
|
||||
}
|
||||
|
||||
auto NewThread = CTX->CreateThread(&NewThreadState, args->args.parent_tid);
|
||||
auto NewThread = CTX->CreateThread(0, 0, &NewThreadState, args->args.parent_tid);
|
||||
CTX->InitializeThread(NewThread);
|
||||
|
||||
if (FEX::HLE::_SyscallHandler->Is64BitMode()) {
|
||||
@@ -131,7 +131,7 @@ namespace FEX::HLE {
|
||||
}
|
||||
|
||||
// Overwrite thread
|
||||
NewThread = CTX->CreateThread(&NewThreadState, GuestArgs->parent_tid);
|
||||
NewThread = CTX->CreateThread(0, 0, &NewThreadState, GuestArgs->parent_tid);
|
||||
|
||||
// CLONE_PARENT_SETTID, CLONE_CHILD_SETTID, CLONE_CHILD_CLEARTID, CLONE_PIDFD will be handled by kernel
|
||||
// Call execution thread directly since we already are on the new thread
|
||||
|
||||
@@ -16,11 +16,10 @@ $end_info$
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <FEXHeaderUtils/ScopedSignalMask.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
|
||||
@@ -55,7 +54,7 @@ bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState *Thread,
|
||||
|
||||
{
|
||||
// Can't use the deferred signal lock in the SIGSEGV handler.
|
||||
FHU::ScopedSignalMaskWithForkableSharedLock lk(_SyscallHandler->VMATracking.Mutex);
|
||||
auto lk = FEXCore::MaskSignalsAndLockMutex<std::shared_lock>(_SyscallHandler->VMATracking.Mutex);
|
||||
|
||||
auto VMATracking = &_SyscallHandler->VMATracking;
|
||||
|
||||
@@ -112,7 +111,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::ScopedDeferredSignalWithForkableSharedLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection<std::shared_lock>(VMATracking.Mutex, Thread);
|
||||
|
||||
// Find the first mapping at or after the range ends, or ::end().
|
||||
// Top points to the address after the end of the range
|
||||
@@ -167,7 +166,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
|
||||
|
||||
// Used for AOT
|
||||
FEXCore::HLE::AOTIRCacheEntryLookupResult SyscallHandler::LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestAddr) {
|
||||
FEXCore::ScopedDeferredSignalWithForkableSharedLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection<std::shared_lock>(VMATracking.Mutex, Thread);
|
||||
|
||||
// Get the first mapping after GuestAddr, or end
|
||||
// GuestAddr is inclusive
|
||||
@@ -194,8 +193,8 @@ void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState *Thread, uintp
|
||||
{
|
||||
// NOTE: Frontend calls this with a nullptr Thread during initialization, but
|
||||
// providing this code with a valid Thread object earlier would allow
|
||||
// us to be more optimal by using ScopedDeferredSignalWithUniqueLock instead
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
// us to be more optimal by using GuardSignalDeferringSection instead
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
|
||||
|
||||
static uint64_t AnonSharedId = 1;
|
||||
|
||||
@@ -244,9 +243,9 @@ void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
|
||||
{
|
||||
// Frontend calls this with nullptr Thread during initialization.
|
||||
// This is why `ScopedPotentialDeferredSignalWithUniqueLock` is used here.
|
||||
// This is why `GuardSignalDeferringSectionWithFallback` is used here.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
|
||||
|
||||
VMATracking.ClearUnsafe(CTX, Base, Size);
|
||||
}
|
||||
@@ -260,7 +259,7 @@ void SyscallHandler::TrackMprotect(FEXCore::Core::InternalThreadState *Thread, u
|
||||
Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
VMATracking.ChangeUnsafe(Base, Size, VMAProt::fromProt(Prot));
|
||||
}
|
||||
@@ -275,7 +274,7 @@ void SyscallHandler::TrackMremap(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
NewSize = FEXCore::AlignUp(NewSize, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
const auto OldVMA = VMATracking.LookupVMAUnsafe(OldAddress);
|
||||
|
||||
@@ -333,7 +332,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState *Thread, int
|
||||
uint64_t Length = stat.shm_segsz;
|
||||
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
// TODO
|
||||
MRID mrid{SpecialDev::SHM, static_cast<uint64_t>(shmid)};
|
||||
@@ -355,7 +354,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState *Thread, int
|
||||
void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState *Thread, uintptr_t Base) {
|
||||
uintptr_t Length = 0;
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
Length = VMATracking.ClearShmUnsafe(CTX, Base);
|
||||
}
|
||||
@@ -369,7 +368,7 @@ void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState *Thread, uint
|
||||
void SyscallHandler::TrackMadvise(FEXCore::Core::InternalThreadState *Thread, uintptr_t Base, uintptr_t Size, int advice) {
|
||||
Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE);
|
||||
{
|
||||
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
// TODO
|
||||
}
|
||||
}
|
||||
|
||||
@@ -245,8 +245,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
|
||||
CTX->InitializeContext();
|
||||
|
||||
#ifndef _WIN32
|
||||
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), {});
|
||||
#else
|
||||
@@ -303,9 +301,9 @@ int main(int argc, char **argv, char **const envp) {
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
|
||||
bool Result1 = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
if (!Result1) {
|
||||
if (!ParentThread) {
|
||||
return 1;
|
||||
}
|
||||
|
||||
@@ -315,7 +313,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
}
|
||||
|
||||
// Just re-use compare state. It also checks against the expected values in config.
|
||||
CTX->GetCPUState(&State);
|
||||
memcpy(&State, &ParentThread->CurrentFrame->State, sizeof(State));
|
||||
|
||||
SyscallHandler.reset();
|
||||
}
|
||||
|
||||
@@ -212,6 +212,7 @@ namespace WorkingAppsTester {
|
||||
|
||||
// EroFS specific
|
||||
static bool Has_EroFSFuse {false};
|
||||
static bool Has_EroFSFsck {false};
|
||||
|
||||
void CheckCurl() {
|
||||
// Check if curl exists on the host
|
||||
@@ -295,11 +296,23 @@ namespace WorkingAppsTester {
|
||||
Has_EroFSFuse = Result != -1;
|
||||
}
|
||||
|
||||
void CheckEroFSFsck() {
|
||||
std::vector<const char*> ExecveArgs = {
|
||||
"fsck.erofs",
|
||||
"-V",
|
||||
nullptr,
|
||||
};
|
||||
|
||||
int32_t Result = Exec::ExecAndWaitForResponseRedirect(ExecveArgs[0], const_cast<char* const*>(ExecveArgs.data()), -1, -1);
|
||||
Has_EroFSFsck = Result != -1;
|
||||
}
|
||||
|
||||
void Init() {
|
||||
CheckCurl();
|
||||
CheckSquashfuse();
|
||||
CheckUnsquashfs();
|
||||
CheckEroFSFuse();
|
||||
CheckEroFSFsck();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -476,12 +489,9 @@ namespace WebFileFetcher {
|
||||
const static std::string DownloadURL = "https://rootfs.fex-emu.gg/RootFS_links.json";
|
||||
|
||||
std::string DownloadToString(const std::string &URL) {
|
||||
std::string BigArgs =
|
||||
fmt::format("curl {}", URL);
|
||||
std::vector<const char*> ExecveArgs = {
|
||||
"/bin/sh",
|
||||
"-c",
|
||||
BigArgs.c_str(),
|
||||
"curl",
|
||||
URL.c_str(),
|
||||
nullptr,
|
||||
};
|
||||
|
||||
@@ -492,12 +502,11 @@ namespace WebFileFetcher {
|
||||
auto filename = URL.substr(URL.find_last_of('/') + 1);
|
||||
auto PathName = Path + filename;
|
||||
|
||||
std::string BigArgs =
|
||||
fmt::format("curl {} -o {}", URL, PathName);
|
||||
std::vector<const char*> ExecveArgs = {
|
||||
"/bin/sh",
|
||||
"-c",
|
||||
BigArgs.c_str(),
|
||||
"curl",
|
||||
URL.c_str(),
|
||||
"-o",
|
||||
PathName.c_str(),
|
||||
nullptr,
|
||||
};
|
||||
|
||||
@@ -1091,7 +1100,7 @@ namespace UnSquash {
|
||||
bool Extract = true;
|
||||
std::error_code ec;
|
||||
if (std::filesystem::exists(TargetFolder, ec)) {
|
||||
fextl::string Question = FolderName + " Already exists. Overwrite?";
|
||||
fextl::string Question = "Target folder \"" + FolderName + "\" already exists. Overwrite?";
|
||||
if (AskForConfirmation(Question)) {
|
||||
if (std::filesystem::remove_all(TargetFolder, ec) != ~0ULL) {
|
||||
Extract = true;
|
||||
@@ -1114,6 +1123,40 @@ namespace UnSquash {
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool ExtractEroFS(const fextl::string &Path, const fextl::string &RootFS, const fextl::string &FolderName) {
|
||||
auto TargetFolder = Path + FolderName;
|
||||
|
||||
bool Extract = true;
|
||||
std::error_code ec;
|
||||
if (std::filesystem::exists(TargetFolder, ec)) {
|
||||
fextl::string Question = "Target folder \"" + FolderName + "\" already exists. Overwrite?";
|
||||
if (AskForConfirmation(Question)) {
|
||||
if (std::filesystem::remove_all(TargetFolder, ec) != ~0ULL) {
|
||||
Extract = true;
|
||||
}
|
||||
if (ec) {
|
||||
ExecWithInfo("Couldn't remove previous directory. Won't extract.");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Extract) {
|
||||
ExecWithInfo("Extracting Erofs. This might take a few minutes.");
|
||||
|
||||
const auto ExtractOption = fmt::format("--extract={}", TargetFolder);
|
||||
const std::vector<const char*> ExecveArgs = {
|
||||
"fsck.erofs",
|
||||
ExtractOption.c_str(),
|
||||
RootFS.c_str(),
|
||||
nullptr,
|
||||
};
|
||||
|
||||
return Exec::ExecAndWaitForResponse(ExecveArgs[0], const_cast<char* const*>(ExecveArgs.data())) == 0;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
int main(int argc, char **argv, char **const envp) {
|
||||
@@ -1240,62 +1283,91 @@ int main(int argc, char **argv, char **const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
struct ExtractStrings {
|
||||
char const *ExtractOrAsIs;
|
||||
char const *AsIsSinceMounterNonFunctional;
|
||||
char const *AsIsSinceExtractorNonFunctional;
|
||||
char const *AsIsSinceNothingWorks;
|
||||
};
|
||||
|
||||
ArgOptions::CompressedImageOption UseImageAs {ArgOptions::CompressedUsageOption};
|
||||
bool HasExtractor{};
|
||||
bool HasMounter{};
|
||||
std::function<bool (const fextl::string &Path, const fextl::string &RootFS, const fextl::string &FolderName)> ExtractHelper;
|
||||
ExtractStrings ExtractingStrings;
|
||||
if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_SQUASHFS) {
|
||||
HasExtractor = WorkingAppsTester::Has_Unsquashfs;
|
||||
HasMounter = WorkingAppsTester::Has_Squashfuse;
|
||||
ExtractHelper = UnSquash::UnsquashRootFS;
|
||||
ExtractingStrings =
|
||||
{
|
||||
"Do you wish to extract the squashfs file or use it as-is?",
|
||||
"Squashfuse doesn't work. Do you wish to extract the squashfs file?",
|
||||
"Unsquashfs doesn't work. Do you want to use the squashfs file as-is?",
|
||||
"Unsquashfs and squashfuse isn't working. Leaving rootfs as-is",
|
||||
};
|
||||
}
|
||||
else if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_EROFS) {
|
||||
HasExtractor = WorkingAppsTester::Has_EroFSFsck;
|
||||
HasMounter = WorkingAppsTester::Has_EroFSFuse;
|
||||
ExtractHelper = UnSquash::ExtractEroFS;
|
||||
ExtractingStrings =
|
||||
{
|
||||
"Do you wish to extract the erofs file or use it as-is?",
|
||||
"erofsfuse doesn't work. Do you wish to extract the erofs file?",
|
||||
"Extracting erofs doesn't work. Do you want to use the erofs file as-is?",
|
||||
"Extracting erofs and erofsfuse isn't working. Leaving rootfs as-is",
|
||||
};
|
||||
}
|
||||
|
||||
int32_t Result{};
|
||||
std::vector<fextl::string> Args = {
|
||||
"Extract",
|
||||
"As-Is",
|
||||
};
|
||||
|
||||
ArgOptions::CompressedImageOption UseImageAs {ArgOptions::CompressedUsageOption};
|
||||
if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_SQUASHFS) {
|
||||
int32_t Result{};
|
||||
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_ASK) {
|
||||
if (WorkingAppsTester::Has_Unsquashfs) {
|
||||
if (WorkingAppsTester::Has_Squashfuse) {
|
||||
Result = AskForConfirmationList("Do you wish to extract the squashfs file or use it as-is?", Args);
|
||||
if (Result == 0) {
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
|
||||
}
|
||||
else if (Result == 1) {
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
|
||||
}
|
||||
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_ASK) {
|
||||
if (HasExtractor) {
|
||||
if (HasMounter) {
|
||||
Result = AskForConfirmationList(ExtractingStrings.ExtractOrAsIs, Args);
|
||||
if (Result == 0) {
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
|
||||
}
|
||||
else {
|
||||
Args.pop_back();
|
||||
Result = AskForConfirmationList("Squashfuse doesn't work. Do you wish to extract the squashfs file?", Args);
|
||||
if (Result == 0) {
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (WorkingAppsTester::Has_Squashfuse) {
|
||||
Args.erase(Args.begin());
|
||||
Result = AskForConfirmationList("Unsquashfs doesn't work. Do you want to use the squashfs file as-is?", Args);
|
||||
if (Result == 0) {
|
||||
// We removed an argument, Just change "As-Is" from 0 to 1 for later logic to work
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
|
||||
}
|
||||
}
|
||||
else {
|
||||
Args.erase(Args.begin());
|
||||
ExecWithInfo("Unsquashfs and squashfuse isn't working. Leaving rootfs as-is");
|
||||
else if (Result == 1) {
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
|
||||
}
|
||||
}
|
||||
else {
|
||||
Args.pop_back();
|
||||
Result = AskForConfirmationList(ExtractingStrings.AsIsSinceMounterNonFunctional, Args);
|
||||
if (Result == 0) {
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_EXTRACT) {
|
||||
auto FolderName = filename.substr(0, filename.find_last_of('.'));
|
||||
if (UnSquash::UnsquashRootFS(RootFS, PathName, FolderName)) {
|
||||
// Remove the .sqsh suffix since we extracted to that
|
||||
filename = FolderName;
|
||||
else {
|
||||
if (HasMounter) {
|
||||
Args.erase(Args.begin());
|
||||
Result = AskForConfirmationList(ExtractingStrings.AsIsSinceExtractorNonFunctional, Args);
|
||||
if (Result == 0) {
|
||||
// We removed an argument, Just change "As-Is" from 0 to 1 for later logic to work
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
|
||||
}
|
||||
}
|
||||
else {
|
||||
Args.erase(Args.begin());
|
||||
ExecWithInfo(ExtractingStrings.AsIsSinceNothingWorks);
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_EROFS) {
|
||||
// Once erofs tooling is available for easy extraction, offer the same settings to extract as squashfs.
|
||||
// Currently this is unavailable, would need a mount + copy + unmount dance.
|
||||
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
|
||||
|
||||
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_EXTRACT) {
|
||||
auto FolderName = filename.substr(0, filename.find_last_of('.'));
|
||||
if (ExtractHelper(RootFS, PathName, FolderName)) {
|
||||
// Remove the image file suffix since we extracted to that.
|
||||
filename = FolderName;
|
||||
}
|
||||
}
|
||||
|
||||
if (AskForConfirmation("Do you wish to set this RootFS as default?")) {
|
||||
|
||||
@@ -22,12 +22,13 @@ EXPORTS
|
||||
Wow64PassExceptionToGuest @16
|
||||
Wow64PrepareForDebuggerAttach @17 PRIVATE
|
||||
Wow64PrepareForException @18
|
||||
Wow64RaiseException @19
|
||||
Wow64ShallowThunkAllocObjectAttributes32TO64_FNC @20 PRIVATE
|
||||
Wow64ShallowThunkAllocSecurityQualityOfService32TO64_FNC @21 PRIVATE
|
||||
Wow64ShallowThunkSIZE_T32TO64 @22 PRIVATE
|
||||
Wow64ShallowThunkSIZE_T64TO32 @23 PRIVATE
|
||||
Wow64SuspendLocalThread @24
|
||||
Wow64SystemServiceEx @25
|
||||
Wow64ValidateUserCallTarget @26 PRIVATE
|
||||
Wow64ValidateUserCallTargetFilter @27 PRIVATE
|
||||
Wow64ProcessPendingCrossProcessItems @19
|
||||
Wow64RaiseException @20
|
||||
Wow64ShallowThunkAllocObjectAttributes32TO64_FNC @21 PRIVATE
|
||||
Wow64ShallowThunkAllocSecurityQualityOfService32TO64_FNC @22 PRIVATE
|
||||
Wow64ShallowThunkSIZE_T32TO64 @23 PRIVATE
|
||||
Wow64ShallowThunkSIZE_T64TO32 @24 PRIVATE
|
||||
Wow64SuspendLocalThread @25 PRIVATE
|
||||
Wow64SystemServiceEx @26
|
||||
Wow64ValidateUserCallTarget @27 PRIVATE
|
||||
Wow64ValidateUserCallTargetFilter @28 PRIVATE
|
||||
@@ -35,6 +35,7 @@ $end_info$
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
#include <utility>
|
||||
#include <unordered_set>
|
||||
#include <ntstatus.h>
|
||||
#include <windef.h>
|
||||
#include <winternl.h>
|
||||
@@ -94,6 +95,7 @@ namespace {
|
||||
SYSTEM_CPU_INFORMATION CpuInfo{};
|
||||
|
||||
std::mutex ThreadSuspendLock;
|
||||
std::unordered_set<DWORD> InitializedWOWThreads; // Set of TIDs, `ThreadSuspendLock` must be locked when accessing
|
||||
|
||||
std::pair<NTSTATUS, TLS> GetThreadTLS(HANDLE Thread) {
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
@@ -179,7 +181,7 @@ namespace Context {
|
||||
Context->Esp = State.gregs[FEXCore::X86State::REG_RSP];
|
||||
|
||||
Context->Eip = State.rip;
|
||||
Context->EFlags = CTX->ReconstructCompactedEFLAGS(Thread);
|
||||
Context->EFlags = CTX->ReconstructCompactedEFLAGS(Thread, false, nullptr, 0);
|
||||
|
||||
Context->SegEs = State.es_idx;
|
||||
Context->SegCs = State.cs_idx;
|
||||
@@ -460,6 +462,7 @@ public:
|
||||
const uint64_t EntryRAX = Frame->State.gregs[FEXCore::X86State::REG_RAX];
|
||||
|
||||
Context::UnlockJITContext();
|
||||
Wow64ProcessPendingCrossProcessItems();
|
||||
ReturnRAX = static_cast<uint64_t>(Wow64SystemServiceEx(static_cast<UINT>(EntryRAX),
|
||||
reinterpret_cast<UINT *>(ReturnRSP + 4)));
|
||||
Context::LockJITContext();
|
||||
@@ -512,7 +515,6 @@ void BTCpuProcessInit() {
|
||||
SyscallHandler = fextl::make_unique<WowSyscallHandler>();
|
||||
|
||||
CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
CTX->InitializeContext();
|
||||
CTX->SetSignalDelegator(SignalDelegator.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
CTX->InitCore(0, 0);
|
||||
@@ -552,8 +554,10 @@ void BTCpuProcessInit() {
|
||||
}
|
||||
|
||||
NTSTATUS BTCpuThreadInit() {
|
||||
GetTLS().ThreadState() = CTX->CreateThread(nullptr, 0);
|
||||
GetTLS().ThreadState() = CTX->CreateThread(0, 0);
|
||||
|
||||
std::scoped_lock Lock(ThreadSuspendLock);
|
||||
InitializedWOWThreads.emplace(GetCurrentThreadId());
|
||||
return STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -563,6 +567,17 @@ NTSTATUS BTCpuThreadTerm(HANDLE Thread) {
|
||||
return Err;
|
||||
}
|
||||
|
||||
{
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
if (NTSTATUS Err = NtQueryInformationThread(Thread, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
|
||||
return Err;
|
||||
}
|
||||
|
||||
const auto ThreadTID = reinterpret_cast<uint64_t>(Info.ClientId.UniqueThread);
|
||||
std::scoped_lock Lock(ThreadSuspendLock);
|
||||
InitializedWOWThreads.erase(ThreadTID);
|
||||
}
|
||||
|
||||
CTX->DestroyThread(TLS.ThreadState());
|
||||
return STATUS_SUCCESS;
|
||||
}
|
||||
@@ -665,6 +680,11 @@ NTSTATUS BTCpuSuspendLocalThread(HANDLE Thread, ULONG *Count) {
|
||||
}
|
||||
|
||||
std::scoped_lock Lock(ThreadSuspendLock);
|
||||
|
||||
// If the thread hasn't yet been initialized, suspend it without special handling as it wont yet have entered the JIT
|
||||
if (!InitializedWOWThreads.contains(ThreadTID))
|
||||
return NtSuspendThread(Thread, Count);
|
||||
|
||||
// If CONTROL_IN_JIT is unset at this point, then it can never be set (and thus the JIT cannot be reentered) as
|
||||
// CONTROL_PAUSED has been set, as such, while this may redundantly request interrupts in rare cases it will never
|
||||
// miss them
|
||||
|
||||
@@ -90,6 +90,7 @@ typedef enum _MEMORY_INFORMATION_CLASS {
|
||||
} MEMORY_INFORMATION_CLASS;
|
||||
|
||||
NTSTATUS WINAPI Wow64SystemServiceEx(UINT,UINT*);
|
||||
void WINAPI Wow64ProcessPendingCrossProcessItems(void);
|
||||
|
||||
NTSTATUS WINAPI RtlWow64SetThreadContext(HANDLE,const WOW64_CONTEXT*);
|
||||
NTSTATUS WINAPI RtlWow64GetThreadContext(HANDLE,WOW64_CONTEXT*);
|
||||
|
||||
@@ -232,14 +232,6 @@ extern "C" {
|
||||
return rv;
|
||||
}
|
||||
|
||||
static void LockMutexFunction(LockInfoPtr) {
|
||||
fprintf(stderr, "libX11: LockMutex\n");
|
||||
}
|
||||
|
||||
static void UnlockMutexFunction(LockInfoPtr) {
|
||||
fprintf(stderr, "libX11: LockMutex\n");
|
||||
}
|
||||
|
||||
int XFree(void* ptr) {
|
||||
// This function must be able to handle both guest heap pointers *and* host heap pointers,
|
||||
// so it only forwards to the native host library for the latter.
|
||||
@@ -368,8 +360,8 @@ extern "C" {
|
||||
return fexfn_pack_XUnregisterIMInstantiateCallback(dpy, rdb, res_name, res_class, AllocateHostTrampolineForGuestFunction(callback), client_data);
|
||||
}
|
||||
|
||||
void (*_XLockMutex_fn)(LockInfoPtr) = LockMutexFunction;
|
||||
void (*_XUnlockMutex_fn)(LockInfoPtr) = UnlockMutexFunction;
|
||||
void (*_XLockMutex_fn)(LockInfoPtr) = nullptr;
|
||||
void (*_XUnlockMutex_fn)(LockInfoPtr) = nullptr;
|
||||
LockInfoPtr _Xglobal_lock = (LockInfoPtr)0x4142434445464748ULL;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# FEX-2311.1
|
||||
# FEX-2312.1
|
||||
|
||||
## FEXCore
|
||||
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
@@ -64,6 +64,10 @@ Metadata that drives the frontend x86/64 decoding
|
||||
- [X87.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp): Handles x86/64 x87 to IR
|
||||
- [X87F64.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp): Handles x86/64 x87 to IR
|
||||
- [OpcodeDispatcher.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_BACKUP_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BACKUP_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_BASE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BASE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_LOCAL_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_LOCAL_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_REMOTE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_REMOTE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
|
||||
|
||||
|
||||
@@ -77,10 +81,6 @@ Logic that binds various parts together
|
||||
Emulation mainloop related glue logic
|
||||
- [Core.cpp](../FEXCore/Source/Interface/Core/Core.cpp): Glues Frontend, OpDispatcher and IR Opts & Compilation, LookupCache, Dispatcher and provides the Execution loop entrypoint
|
||||
|
||||
#### gdbserver
|
||||
- [GdbServer.cpp](../FEXCore/Source/Interface/Core/GdbServer.cpp): Provides a gdb interface to the guest state
|
||||
- [GdbServer.h](../FEXCore/Source/Interface/Core/GdbServer.h)
|
||||
|
||||
#### log-manager
|
||||
- [LogManager.cpp](../FEXCore/Source/Utils/LogManager.cpp)
|
||||
|
||||
@@ -143,6 +143,10 @@ Text -> IR
|
||||
- [X87.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp): Handles x86/64 x87 to IR
|
||||
- [X87F64.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp): Handles x86/64 x87 to IR
|
||||
- [OpcodeDispatcher.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_BACKUP_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BACKUP_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_BASE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BASE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_LOCAL_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_LOCAL_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
- [OpcodeDispatcher_REMOTE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_REMOTE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
|
||||
## ThunkLibs
|
||||
See [ThunkLibs/README.md](../ThunkLibs/README.md) for more details
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x000000007dbf2800",
|
||||
"RDX": "0x0000000000000000",
|
||||
"RBX": "0x000000000000004f",
|
||||
"RCX": "0x000000000000004f",
|
||||
"RBP": "0x0000000000009e4f",
|
||||
"RSI": "0x0000000000009e4f",
|
||||
"RSP": "0x000000000000004f"
|
||||
},
|
||||
"Mode": "32BIT"
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug where smaller than 64-bit imul could leave garbage data in the upper 32-bits of the 32-bit result.
|
||||
; This would cause subsequent instructions after the imul to receive garbage bits.
|
||||
; In particular this would feed in to address calculation in DXVK with "Dungeon Defenders" doing address calculation.
|
||||
; The address calculation did something similar to:
|
||||
; xor edx, edx
|
||||
; mov eax, 0x7dbf2800
|
||||
; imul ebx, ebx, 0xaaaaaaab
|
||||
; div ebx
|
||||
; Divide expected 0x4f but received 0xffffffb1'0000'004f
|
||||
|
||||
; Dividend
|
||||
xor edx, edx
|
||||
mov eax, 0x7dbf2800
|
||||
|
||||
; Multiply starting value
|
||||
mov ebx, 0xED
|
||||
|
||||
jmp .test
|
||||
|
||||
.test:
|
||||
|
||||
; imul 1-src
|
||||
mov edi, 0xaaaaaaab
|
||||
imul di, bx
|
||||
mov esp, 0xaaaaaaab
|
||||
imul esp, ebx
|
||||
|
||||
; imul 2-src 8-bit check
|
||||
imul bp, bx, 0xab
|
||||
imul esi, ebx, 0xab
|
||||
|
||||
; imul 2-src 16-bit check
|
||||
imul cx, bx, 0xaaab
|
||||
imul ebx, ebx, 0xaaaaaaab
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,45 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4",
|
||||
"RBX": "0xFFFFFFFFFFFFFFF4",
|
||||
"RCX": "0x0",
|
||||
"RDX": "0x1337"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug where bzhi would fail to update SF. Test that bzhi correctly
|
||||
; sets ZF/SF correctly based on the result.
|
||||
|
||||
mov rcx, 4
|
||||
mov rbx, -12
|
||||
|
||||
; Result is 0x4
|
||||
bzhi rax, rbx, rcx
|
||||
mov rdx, 0xdead1
|
||||
jz .fail
|
||||
mov rdx, 0xdead2
|
||||
js .fail
|
||||
|
||||
; Result is -12
|
||||
mov rcx, 64
|
||||
bzhi rdx, rbx, rcx
|
||||
mov rdx, 0xdead3
|
||||
jz .fail
|
||||
mov rdx, 0xdead4
|
||||
jns .fail
|
||||
|
||||
; Result is 0x00
|
||||
mov rdx, 0
|
||||
bzhi rcx, rbx, rdx
|
||||
mov rdx, 0xdead5
|
||||
jnz .fail
|
||||
mov rdx, 0xdead6
|
||||
js .fail
|
||||
|
||||
mov rdx, 0x1337
|
||||
hlt
|
||||
|
||||
.fail:
|
||||
hlt
|
||||
@@ -0,0 +1,15 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x500000020"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug in its `TestNZ` opcode where it would try to load a constant in to the tst instruction
|
||||
; If the constant didn't fit in a logical encoding it would generate invalid instructions and also crash.
|
||||
; This snippet of code was found in libGLX.so.0.0.0 when trying to load steamwebhelper.
|
||||
mov eax, 0x28000001
|
||||
shl rax, 0x5
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,49 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0",
|
||||
"XMM0": ["0", "0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX-Emu has a bug around NZCV flags getting spilled and filled.
|
||||
; The bug comes down to NZCV actually being 32-bit but our IR incorrectly assumed that all flags were 8-bit.
|
||||
; Once a spill situation happened, it would only store and reload the lower 8-bits of the NZCV flag which wasn't correct.
|
||||
; This caused this code to infinite loop and read past memory and crash.
|
||||
|
||||
; Code found from Ender Lilies in their `sha1_block_data_order` function which is significantly longer than this snippit.
|
||||
lea rsi, [rel .data_vecs]
|
||||
mov rax, 1
|
||||
|
||||
; Break visibility
|
||||
jmp loop_top
|
||||
loop_top:
|
||||
|
||||
; Decrement counter.
|
||||
dec rax
|
||||
|
||||
; Load rsi + 0x40 in to rbx
|
||||
lea rbx, [rsi+0x40]
|
||||
|
||||
; Move rbx in to rsi, incrementing the pointer by 64-bytes if rax isn't zero.
|
||||
cmovne rsi, rbx
|
||||
|
||||
; Do a sha1rnds4, which uses enough temporaries to spill NZCV which picks up a crash.
|
||||
sha1rnds4 xmm0, xmm0, 0x0
|
||||
|
||||
; This memory access will crash once we loop too many times.
|
||||
movdqu xmm0, [rsi]
|
||||
|
||||
; Jump back to the top
|
||||
jne loop_top
|
||||
|
||||
hlt
|
||||
|
||||
.data_vecs:
|
||||
dq 0, 0, 0, 0
|
||||
dq 0, 0, 0, 0
|
||||
dq 0, 0, 0, 0
|
||||
dq 0, 0, 0, 0
|
||||
dq 0, 0, 0, 0
|
||||
dq 0, 0, 0, 0
|
||||
@@ -12,7 +12,6 @@
|
||||
"Instructions": {
|
||||
"roundss xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Nearest rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
@@ -23,7 +22,6 @@
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
@@ -34,7 +32,6 @@
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
@@ -45,7 +42,6 @@
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
@@ -56,7 +52,6 @@
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host rounding mode rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
@@ -67,7 +62,6 @@
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Nearest rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
@@ -78,7 +72,6 @@
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
@@ -89,7 +82,6 @@
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
@@ -100,7 +92,6 @@
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
@@ -111,7 +102,6 @@
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host rounding mode rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
"Instructions": {
|
||||
"cvtpi2ps xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
@@ -22,12 +21,11 @@
|
||||
},
|
||||
"cvtpi2ps xmm0, mm0": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d2, [x28, #768]",
|
||||
"scvtf v16.2s, v2.2s"
|
||||
]
|
||||
}
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
"Instructions": {
|
||||
"cvtsi2ss xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
@@ -21,7 +20,6 @@
|
||||
},
|
||||
"cvtsi2ss xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
@@ -32,7 +30,6 @@
|
||||
},
|
||||
"cvtsi2ss xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
@@ -43,7 +40,6 @@
|
||||
},
|
||||
"sqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x51",
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt s16, s17"
|
||||
@@ -51,7 +47,6 @@
|
||||
},
|
||||
"rsqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x52"
|
||||
@@ -64,7 +59,6 @@
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x53"
|
||||
@@ -76,7 +70,6 @@
|
||||
},
|
||||
"addss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x58"
|
||||
],
|
||||
@@ -86,7 +79,6 @@
|
||||
},
|
||||
"mulss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x59"
|
||||
],
|
||||
@@ -96,7 +88,6 @@
|
||||
},
|
||||
"cvtss2sd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt d16, s17"
|
||||
@@ -104,7 +95,6 @@
|
||||
},
|
||||
"cvtss2sd xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
@@ -113,7 +103,6 @@
|
||||
},
|
||||
"subss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5c"
|
||||
],
|
||||
@@ -123,7 +112,6 @@
|
||||
},
|
||||
"minss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5d"
|
||||
],
|
||||
@@ -133,7 +121,6 @@
|
||||
},
|
||||
"divss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5e"
|
||||
],
|
||||
@@ -143,7 +130,6 @@
|
||||
},
|
||||
"maxss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5f"
|
||||
],
|
||||
@@ -153,7 +139,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -163,7 +148,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -173,7 +157,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -183,7 +166,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -198,7 +180,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -211,7 +192,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -223,7 +203,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -235,7 +214,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
"Instructions": {
|
||||
"cvtsi2sd xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -21,7 +20,6 @@
|
||||
},
|
||||
"cvtsi2sd xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -32,7 +30,6 @@
|
||||
},
|
||||
"cvtsi2sd xmm0, rax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -42,7 +39,6 @@
|
||||
},
|
||||
"cvtsi2sd xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -53,7 +49,6 @@
|
||||
},
|
||||
"sqrtsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x51"
|
||||
],
|
||||
@@ -63,7 +58,6 @@
|
||||
},
|
||||
"addsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x58"
|
||||
],
|
||||
@@ -73,7 +67,6 @@
|
||||
},
|
||||
"mulsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x59"
|
||||
],
|
||||
@@ -83,7 +76,6 @@
|
||||
},
|
||||
"cvtsd2ss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
@@ -93,7 +85,6 @@
|
||||
},
|
||||
"cvtsd2ss xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
@@ -104,7 +95,6 @@
|
||||
},
|
||||
"subsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5c"
|
||||
],
|
||||
@@ -114,7 +104,6 @@
|
||||
},
|
||||
"minsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5d"
|
||||
],
|
||||
@@ -124,7 +113,6 @@
|
||||
},
|
||||
"divsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5e"
|
||||
],
|
||||
@@ -134,7 +122,6 @@
|
||||
},
|
||||
"maxsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5f"
|
||||
],
|
||||
@@ -144,7 +131,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -154,7 +140,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -164,7 +149,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -174,7 +158,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -189,7 +172,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -202,7 +184,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -214,7 +195,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -226,7 +206,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
"Instructions": {
|
||||
"cvtpi2ps xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
@@ -23,12 +22,11 @@
|
||||
},
|
||||
"cvtpi2ps xmm0, mm0": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d2, [x28, #768]",
|
||||
"scvtf v16.2s, v2.2s"
|
||||
]
|
||||
}
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
"Instructions": {
|
||||
"cvtsi2ss xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
@@ -22,7 +21,6 @@
|
||||
},
|
||||
"cvtsi2ss xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
@@ -33,7 +31,6 @@
|
||||
},
|
||||
"cvtsi2ss xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
@@ -44,7 +41,6 @@
|
||||
},
|
||||
"sqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x51",
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt s16, s17"
|
||||
@@ -52,7 +48,6 @@
|
||||
},
|
||||
"rsqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x52"
|
||||
@@ -65,7 +60,6 @@
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x53"
|
||||
@@ -77,7 +71,6 @@
|
||||
},
|
||||
"addss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x58"
|
||||
],
|
||||
@@ -87,7 +80,6 @@
|
||||
},
|
||||
"mulss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x59"
|
||||
],
|
||||
@@ -97,7 +89,6 @@
|
||||
},
|
||||
"cvtss2sd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt d16, s17"
|
||||
@@ -105,7 +96,6 @@
|
||||
},
|
||||
"cvtss2sd xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
@@ -114,7 +104,6 @@
|
||||
},
|
||||
"subss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5c"
|
||||
],
|
||||
@@ -124,7 +113,6 @@
|
||||
},
|
||||
"minss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5d"
|
||||
],
|
||||
@@ -134,7 +122,6 @@
|
||||
},
|
||||
"divss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5e"
|
||||
],
|
||||
@@ -144,7 +131,6 @@
|
||||
},
|
||||
"maxss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5f"
|
||||
],
|
||||
@@ -154,7 +140,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -164,7 +149,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -174,7 +158,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -184,7 +167,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -198,7 +180,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -210,7 +191,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -222,7 +202,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
@@ -234,7 +213,6 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
"Instructions": {
|
||||
"cvtsi2sd xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -22,7 +21,6 @@
|
||||
},
|
||||
"cvtsi2sd xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -33,7 +31,6 @@
|
||||
},
|
||||
"cvtsi2sd xmm0, rax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -43,7 +40,6 @@
|
||||
},
|
||||
"cvtsi2sd xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
@@ -54,7 +50,6 @@
|
||||
},
|
||||
"sqrtsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x51"
|
||||
],
|
||||
@@ -64,7 +59,6 @@
|
||||
},
|
||||
"addsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x58"
|
||||
],
|
||||
@@ -74,7 +68,6 @@
|
||||
},
|
||||
"mulsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x59"
|
||||
],
|
||||
@@ -84,7 +77,6 @@
|
||||
},
|
||||
"cvtsd2ss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
@@ -94,7 +86,6 @@
|
||||
},
|
||||
"cvtsd2ss xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
@@ -105,7 +96,6 @@
|
||||
},
|
||||
"subsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5c"
|
||||
],
|
||||
@@ -115,7 +105,6 @@
|
||||
},
|
||||
"minsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5d"
|
||||
],
|
||||
@@ -125,7 +114,6 @@
|
||||
},
|
||||
"divsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5e"
|
||||
],
|
||||
@@ -135,7 +123,6 @@
|
||||
},
|
||||
"maxsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5f"
|
||||
],
|
||||
@@ -145,7 +132,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -155,7 +141,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -165,7 +150,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -175,7 +159,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -189,7 +172,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -201,7 +183,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -213,7 +194,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
@@ -225,7 +205,6 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
"Instructions": {
|
||||
"vsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x51 128-bit"
|
||||
],
|
||||
@@ -22,7 +21,6 @@
|
||||
},
|
||||
"vsqrtsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x51 128-bit"
|
||||
],
|
||||
@@ -33,7 +31,6 @@
|
||||
},
|
||||
"vrsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"Map 1 0b10 0x52 128-bit"
|
||||
@@ -47,7 +44,6 @@
|
||||
},
|
||||
"vrcpss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"Map 1 0b10 0x53 128-bit"
|
||||
@@ -60,7 +56,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x00": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -71,7 +66,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x01": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -82,7 +76,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x02": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -93,7 +86,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x03": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -108,7 +100,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x04": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -121,7 +112,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x05": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -134,7 +124,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x06": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -147,7 +136,6 @@
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x07": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
@@ -161,7 +149,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x00": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -172,7 +159,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x01": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -183,7 +169,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x02": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -194,7 +179,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x03": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -209,7 +193,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x04": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -222,7 +205,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x05": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -235,7 +217,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x06": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -248,7 +229,6 @@
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x07": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
@@ -262,7 +242,6 @@
|
||||
},
|
||||
"vcvtsi2ss xmm0, xmm1, eax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x2A 128-bit"
|
||||
],
|
||||
@@ -273,7 +252,6 @@
|
||||
},
|
||||
"vcvtsi2ss xmm0, xmm1, rax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x2A 128-bit"
|
||||
],
|
||||
@@ -284,7 +262,6 @@
|
||||
},
|
||||
"vcvtsi2sd xmm0, xmm1, eax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x2A 128-bit"
|
||||
],
|
||||
@@ -295,7 +272,6 @@
|
||||
},
|
||||
"vcvtsi2sd xmm0, xmm1, rax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x2A 128-bit"
|
||||
],
|
||||
@@ -306,7 +282,6 @@
|
||||
},
|
||||
"vmulss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x59 128-bit"
|
||||
],
|
||||
@@ -317,7 +292,6 @@
|
||||
},
|
||||
"vmulsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x59 128-bit"
|
||||
],
|
||||
@@ -328,7 +302,6 @@
|
||||
},
|
||||
"vcvtss2sd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5a 128-bit"
|
||||
],
|
||||
@@ -339,7 +312,6 @@
|
||||
},
|
||||
"vcvtsd2ss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5a 128-bit"
|
||||
],
|
||||
@@ -350,7 +322,6 @@
|
||||
},
|
||||
"vsubss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5c 128-bit"
|
||||
],
|
||||
@@ -361,7 +332,6 @@
|
||||
},
|
||||
"vsubsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5c 128-bit"
|
||||
],
|
||||
@@ -372,7 +342,6 @@
|
||||
},
|
||||
"vminss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5d 128-bit"
|
||||
],
|
||||
@@ -383,7 +352,6 @@
|
||||
},
|
||||
"vminsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5d 128-bit"
|
||||
],
|
||||
@@ -394,7 +362,6 @@
|
||||
},
|
||||
"vdivss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5e 128-bit"
|
||||
],
|
||||
@@ -405,7 +372,6 @@
|
||||
},
|
||||
"vdivsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5e 128-bit"
|
||||
],
|
||||
@@ -416,7 +382,6 @@
|
||||
},
|
||||
"vmaxss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5f 128-bit"
|
||||
],
|
||||
@@ -427,7 +392,6 @@
|
||||
},
|
||||
"vmaxsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5f 128-bit"
|
||||
],
|
||||
@@ -438,7 +402,6 @@
|
||||
},
|
||||
"vminps xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0x5d 128-bit"
|
||||
],
|
||||
@@ -450,7 +413,6 @@
|
||||
},
|
||||
"vminps ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0x5d 256-bit"
|
||||
],
|
||||
@@ -464,7 +426,6 @@
|
||||
},
|
||||
"vminpd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b01 0x5d 128-bit"
|
||||
],
|
||||
@@ -476,7 +437,6 @@
|
||||
},
|
||||
"vminpd ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b01 0x5d 256-bit"
|
||||
],
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
"Instructions": {
|
||||
"vroundss xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"nearest rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
@@ -23,7 +22,6 @@
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
@@ -35,7 +33,6 @@
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
@@ -47,7 +44,6 @@
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
@@ -59,7 +55,6 @@
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host mode rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
@@ -71,7 +66,6 @@
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"nearest rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
@@ -83,7 +77,6 @@
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
@@ -95,7 +88,6 @@
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
@@ -107,7 +99,6 @@
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
@@ -119,7 +110,6 @@
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host mode rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,138 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"CRYPTO"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"sha1nexte xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xc8"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"dup v2.4s, v16.s[3]",
|
||||
"unimplemented (Unimplemented)",
|
||||
"dup v2.4s, v2.s[0]",
|
||||
"add v2.4s, v17.4s, v2.4s",
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.s[3], v2.s[3]"
|
||||
]
|
||||
},
|
||||
"sha256msg1 xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xcc"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"unimplemented (Unimplemented)"
|
||||
]
|
||||
},
|
||||
"aesimc xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xdb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"unimplemented (Unimplemented)"
|
||||
]
|
||||
},
|
||||
"aesenc xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xdc"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"unimplemented (Unimplemented)",
|
||||
"unimplemented (Unimplemented)",
|
||||
"eor v16.16b, v16.16b, v17.16b"
|
||||
]
|
||||
},
|
||||
"aesenclast xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xdd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"unimplemented (Unimplemented)",
|
||||
"eor v16.16b, v16.16b, v17.16b"
|
||||
]
|
||||
},
|
||||
"aesdec xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xde"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"unimplemented (Unimplemented)",
|
||||
"unimplemented (Unimplemented)",
|
||||
"eor v16.16b, v16.16b, v17.16b"
|
||||
]
|
||||
},
|
||||
"aesdeclast xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x38 0xdf"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"unimplemented (Unimplemented)",
|
||||
"eor v16.16b, v16.16b, v17.16b"
|
||||
]
|
||||
},
|
||||
"crc32 eax, bl": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x38 0xf0"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"crc32cb w4, w4, w7"
|
||||
]
|
||||
},
|
||||
"crc32 eax, bx": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x38 0xf1"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"crc32ch w4, w4, w7"
|
||||
]
|
||||
},
|
||||
"crc32 eax, ebx": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x38 0xf1"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"crc32cw w4, w4, w7"
|
||||
]
|
||||
},
|
||||
"crc32 rax, bl": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x38 0xf0"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"crc32cb w4, w4, w7"
|
||||
]
|
||||
},
|
||||
"crc32 rax, rbx": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x38 0xf1"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"crc32cx w4, w4, x7"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"CRYPTO"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"pclmulqdq xmm0, xmm1, 00000b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0x44"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"unallocated (Unallocated)"
|
||||
]
|
||||
},
|
||||
"pclmulqdq xmm0, xmm1, 00001b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0x44"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"dup v0.2d, v16.d[1]",
|
||||
"unallocated (Unallocated)"
|
||||
]
|
||||
},
|
||||
"pclmulqdq xmm0, xmm1, 10000b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0x44"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"dup v0.2d, v17.d[1]",
|
||||
"unallocated (Unallocated)"
|
||||
]
|
||||
},
|
||||
"pclmulqdq xmm0, xmm1, 10001b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0x44"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"unallocated (Unallocated)"
|
||||
]
|
||||
},
|
||||
"aeskeygenassist xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0xdf"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #2080]",
|
||||
"movi v3.2d, #0x0",
|
||||
"mov v16.16b, v17.16b",
|
||||
"unimplemented (Unimplemented)",
|
||||
"tbl v16.16b, {v16.16b}, v2.16b"
|
||||
]
|
||||
},
|
||||
"aeskeygenassist xmm0, xmm1, 0xFF": {
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0xdf"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #2080]",
|
||||
"movi v3.2d, #0x0",
|
||||
"mov v16.16b, v17.16b",
|
||||
"unimplemented (Unimplemented)",
|
||||
"tbl v16.16b, {v16.16b}, v2.16b",
|
||||
"mov x0, #0xff00000000",
|
||||
"dup v1.2d, x0",
|
||||
"eor v16.16b, v16.16b, v1.16b"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -16,322 +16,294 @@
|
||||
"Instructions": {
|
||||
"pi2fw mm0, mm1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x0c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"uzp1 v2.4h, v2.4h, v2.4h",
|
||||
"sxtl v2.4s, v2.4h",
|
||||
"scvtf v2.2s, v2.2s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pi2fd mm0, mm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x0d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"scvtf v2.2s, v2.2s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pf2iw mm0, mm1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x1c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"fcvtzs v2.2s, v2.2s",
|
||||
"uzp1 v2.4h, v2.4h, v2.4h",
|
||||
"sxtl v2.4s, v2.4h",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pf2id mm0, mm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x1d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"fcvtzs v2.2s, v2.2s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrcpv mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x86"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"fmov v0.4s, #0x70 (1.0000)",
|
||||
"fdiv v2.4s, v0.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrsqrtv mm0, mm1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x87"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"fmov v0.4s, #0x70 (1.0000)",
|
||||
"fsqrt v1.4s, v2.4s",
|
||||
"fdiv v2.4s, v0.4s, v1.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfnacc mm0, mm1": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0x8a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #784]",
|
||||
"uzp1 v4.2s, v2.2s, v3.2s",
|
||||
"uzp2 v2.2s, v2.2s, v3.2s",
|
||||
"fsub v2.4s, v4.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfpnacc mm0, mm1": {
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0x8e",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #784]",
|
||||
"dup v4.2s, v2.s[1]",
|
||||
"fsub s2, s2, s4",
|
||||
"faddp v3.4s, v3.4s, v3.4s",
|
||||
"mov v2.s[1], v3.s[0]",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfcmpge mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0x90",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fcmge v2.4s, v3.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfmin mm0, mm1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0x94",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fcmgt v0.4s, v3.4s, v2.4s",
|
||||
"bif v2.16b, v3.16b, v0.16b",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrcp mm0, mm1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x96"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fdiv s2, s0, s2",
|
||||
"dup v2.2s, v2.s[0]",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrsqrt mm0, mm1": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x0f 0x97"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fsqrt s1, s2",
|
||||
"fdiv s2, s0, s1",
|
||||
"dup v2.2s, v2.s[0]",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfsub mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0x9a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fsub v2.4s, v3.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfadd mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0x9e",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fadd v2.4s, v3.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfcmpgt mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xa0",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fcmgt v2.4s, v3.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfmax mm0, mm1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xa4",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fcmgt v0.4s, v3.4s, v2.4s",
|
||||
"bit v2.16b, v3.16b, v0.16b",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrcpit1 mm0, mm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xa6",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"str d2, [x28, #752]"
|
||||
"ldr d2, [x28, #784]",
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrcpit1 mm0, mm0": {
|
||||
"ExpectedInstructionCount": 0,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xa6",
|
||||
"ExpectedArm64ASM": []
|
||||
},
|
||||
"pfrsqit1 mm0, mm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xa7",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"str d2, [x28, #752]"
|
||||
"ldr d2, [x28, #784]",
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrsqit1 mm0, mm0": {
|
||||
"ExpectedInstructionCount": 0,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xa7",
|
||||
"ExpectedArm64ASM": []
|
||||
},
|
||||
"pfsubr mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xaa",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fsub v2.4s, v2.4s, v3.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfcmpeq mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xb0",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fcmeq v2.4s, v3.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfmul mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xb4",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"fmul v2.4s, v3.4s, v2.4s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrcpit2 mm0, mm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xb6",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"str d2, [x28, #752]"
|
||||
"ldr d2, [x28, #784]",
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pfrcpit2 mm0, mm0": {
|
||||
"ExpectedInstructionCount": 0,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xb6",
|
||||
"ExpectedArm64ASM": []
|
||||
},
|
||||
"db 0x0f, 0x0f, 0xc1, 0xb7": {
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"nasm doesn't support emitting this instruction",
|
||||
"pmulhrw mm0, mm1",
|
||||
"0x0f 0x0f 0xb7"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #784]",
|
||||
"smull v2.4s, v2.4h, v3.4h",
|
||||
"movi v3.4s, #0x80, lsl #8",
|
||||
"add v2.4s, v2.4s, v3.4s",
|
||||
"shrn v2.4h, v2.4s, #16",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pswapd mm0, mm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xbb",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"rev64 v2.2s, v2.2s",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
},
|
||||
"pavgusb mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x0f 0xbf",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ldr d2, [x28, #784]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"urhadd v2.16b, v3.16b, v2.16b",
|
||||
"str d2, [x28, #752]"
|
||||
"str d2, [x28, #768]"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,7 +15,6 @@
|
||||
"Instructions": {
|
||||
"push ax, bx": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Mergable 16-bit pushes. May or may not be an optimization."
|
||||
],
|
||||
@@ -30,7 +29,6 @@
|
||||
},
|
||||
"push rax, rbx": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Mergable 64-bit pushes"
|
||||
],
|
||||
@@ -45,7 +43,6 @@
|
||||
},
|
||||
"adds xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Redundant scalar adds that can get eliminated without AFP."
|
||||
],
|
||||
@@ -62,7 +59,6 @@
|
||||
},
|
||||
"positive movsb": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -82,7 +78,6 @@
|
||||
},
|
||||
"positive movsw": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -102,7 +97,6 @@
|
||||
},
|
||||
"positive movsd": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -122,7 +116,6 @@
|
||||
},
|
||||
"positive movsq": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -142,7 +135,6 @@
|
||||
},
|
||||
"negative movsb": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -162,7 +154,6 @@
|
||||
},
|
||||
"negative movsw": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -182,7 +173,6 @@
|
||||
},
|
||||
"negative movsd": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -202,7 +192,6 @@
|
||||
},
|
||||
"negative movsq": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
@@ -219,6 +208,444 @@
|
||||
"sub x10, x10, #0x8 (8)",
|
||||
"sub x11, x11, #0x8 (8)"
|
||||
]
|
||||
},
|
||||
"positive rep movsb": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep movsb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldrb w3, [x2], #1",
|
||||
"strb w3, [x1], #1",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"add x22, x0, x2",
|
||||
"add x23, x1, x2",
|
||||
"mov x11, x22",
|
||||
"mov x10, x23",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"positive rep movsw": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep movsw"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldrh w3, [x2], #2",
|
||||
"strh w3, [x1], #2",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"add x22, x0, x2, lsl #1",
|
||||
"add x23, x1, x2, lsl #1",
|
||||
"mov x11, x22",
|
||||
"mov x10, x23",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"positive rep movsd": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep movsd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldr w3, [x2], #4",
|
||||
"str w3, [x1], #4",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"add x22, x0, x2, lsl #2",
|
||||
"add x23, x1, x2, lsl #2",
|
||||
"mov x11, x22",
|
||||
"mov x10, x23",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"positive rep movsq": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep movsq"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldr x3, [x2], #8",
|
||||
"str x3, [x1], #8",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"add x22, x0, x2, lsl #3",
|
||||
"add x23, x1, x2, lsl #3",
|
||||
"mov x11, x22",
|
||||
"mov x10, x23",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"negative rep movsb": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep movsb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldrb w3, [x2], #-1",
|
||||
"strb w3, [x1], #-1",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"sub x20, x0, x2",
|
||||
"sub x21, x1, x2",
|
||||
"mov x11, x20",
|
||||
"mov x10, x21",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"negative rep movsw": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep movsw"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldrh w3, [x2], #-2",
|
||||
"strh w3, [x1], #-2",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"sub x20, x0, x2, lsl #1",
|
||||
"sub x21, x1, x2, lsl #1",
|
||||
"mov x11, x20",
|
||||
"mov x10, x21",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"negative rep movsd": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep movsd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldr w3, [x2], #-4",
|
||||
"str w3, [x1], #-4",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"sub x20, x0, x2, lsl #2",
|
||||
"sub x21, x1, x2, lsl #2",
|
||||
"mov x11, x20",
|
||||
"mov x10, x21",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"negative rep movsq": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep movsq"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"mov x2, x10",
|
||||
"cbz x0, #+0x14",
|
||||
"ldr x3, [x2], #-8",
|
||||
"str x3, [x1], #-8",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0xc",
|
||||
"mov x0, x11",
|
||||
"mov x1, x10",
|
||||
"mov x2, x5",
|
||||
"sub x20, x0, x2, lsl #3",
|
||||
"sub x21, x1, x2, lsl #3",
|
||||
"mov x11, x20",
|
||||
"mov x10, x21",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"positive rep stosb": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep stosb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"uxtb w21, w4",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"strb w21, [x1], #1",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x5",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"positive rep stosw": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep stosw"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"uxth w21, w4",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"strh w21, [x1], #2",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x5, lsl #1",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"positive rep stosd": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep stosd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov w21, w4",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"str w21, [x1], #4",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x5, lsl #2",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"positive rep stosq": {
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"cld",
|
||||
"rep stosq"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x0",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"str x4, [x1], #8",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"add x11, x11, x5, lsl #3",
|
||||
"mov x5, x20"
|
||||
]
|
||||
},
|
||||
"negative rep stosb": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep stosb"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"uxtb w20, w4",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"strb w20, [x1], #-1",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"sub x11, x11, x5",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"negative rep stosw": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep stosw"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"uxth w20, w4",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"strh w20, [x1], #-2",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"sub x11, x11, x5, lsl #1",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"negative rep stosd": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep stosd"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov w20, w4",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"str w20, [x1], #-4",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"sub x11, x11, x5, lsl #2",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
},
|
||||
"negative rep stosq": {
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Comment": [
|
||||
"When direction flag is a compile time constant we can optimize",
|
||||
"loads and stores can turn in to post-increment when known"
|
||||
],
|
||||
"x86Insts": [
|
||||
"std",
|
||||
"rep stosq"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"strb w20, [x28, #714]",
|
||||
"mov x0, x5",
|
||||
"mov x1, x11",
|
||||
"cbz x0, #+0x10",
|
||||
"str x4, [x1], #-8",
|
||||
"sub x0, x0, #0x1 (1)",
|
||||
"cbnz x0, #-0x8",
|
||||
"sub x11, x11, x5, lsl #3",
|
||||
"mov w5, #0x0"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -16,7 +16,6 @@
|
||||
"Instructions": {
|
||||
"adds xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Redundant scalar operations should get eliminated with AFP"
|
||||
],
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
Loaded 100 of 145 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user